Coverage for scanpath_studio/app.py: 89%
3113 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Scanpath Studio Streamlit app.
3This is the main entry point for the Streamlit application that visualizes
4eye-tracking scanpaths over text.
6Architecture:
7 - Entry point: main() function configures Streamlit and orchestrates the UI
8 - Data flow: CSV upload → schema inference → normalization → filtering → plotting
9 - UI structure: Sidebar controls + views (Scanpath Visualization [with
10 Comparisons + Line assignment subtabs], Corpus Analysis, Data Inspection)
12Data Pipeline:
13 1. Load raw CSVs (words + fixations + optional raw gaze)
14 2. Infer schema via candidate column matching
15 3. Normalize to canonical column names
16 4. Apply participant/trial/text filters
17 5. Build trial combinations for selection
18 6. Render visualizations with user-controlled settings
20Usage:
21 # Development mode (watch for changes):
22 $ streamlit run scanpath_studio/app.py
24 # Package mode:
25 $ python -m scanpath_studio
26 # or
27 $ scanpath-studio
28"""
30from __future__ import annotations
32import base64
33import contextlib
34import copy
35import functools
36import hashlib
37import html
38import json
39import logging
40import os
41import re
42import shutil
43import subprocess
44import sys
45import time
46from collections.abc import Callable, Iterable, Mapping
47from dataclasses import dataclass, replace
48from functools import partial
49from pathlib import Path
50from typing import TYPE_CHECKING
52import pandas as pd
53import streamlit as st
54from streamlit import runtime
55from streamlit.errors import StreamlitAPIException
57# Allow running via `streamlit run scanpath_studio/app.py` by adding the
58# repository root to sys.path when executed as a script instead of a package.
59if __package__ is None or __package__ == "":
60 import sys
61 from pathlib import Path
63 root = Path(__file__).resolve().parent.parent
64 if str(root) not in sys.path:
65 sys.path.insert(0, str(root))
67from scanpath_studio import (
68 dataset_table,
69 desktop_update,
70 loading,
71 progress,
72 wizard_shell,
73)
74from scanpath_studio import metadata as metadata_mod
75from scanpath_studio import updates as update_check
76from scanpath_studio.annotations import (
77 filter_keys,
78)
79from scanpath_studio.column_names import (
80 ACTIVE_COLUMN_NAMES_KEY,
81 ColumnNames,
82 active_all,
83 from_schema,
84)
85from scanpath_studio.column_names import active as active_names
86from scanpath_studio.constants import (
87 _VIEW_CORPUS,
88 _VIEW_DATA,
89 _VIEW_SCANPATH,
90 AUTHOR_CHOICE,
91 BACKGROUND_PRESETS,
92 BENCHMARK_LABEL_SUFFIX,
93 BENCHMARK_SHORT_SUFFIX,
94 BENCHMARK_WIP_SUFFIX,
95 CITATION,
96 DATA_EDITOR_KEY,
97 DATA_EDITOR_OFFSCREEN_KEY,
98 DATA_OVERVIEW_KEY,
99 DATA_PAGE_KEY,
100 DATA_PAGE_OFFSCREEN_KEY,
101 DATASET_ADDED_KEY,
102 DATASET_COUNTS_STORE_KEY,
103 DATASET_DESCRIPTIONS_KEY,
104 DATASET_EDITOR_OPEN_KEY,
105 DATASET_SETUP_OVERRIDES_KEY,
106 DEFAULT_BACKGROUND_COLOR,
107 DEFAULT_FIGURE_SIZE,
108 DEFAULT_LINE_SPACING,
109 DEMO_CHOICE,
110 DOWNLOAD_DIR_ENV,
111 DOWNLOAD_DIR_KEY,
112 EYEGENBENCH_DEFAULT_DIR,
113 FOCUS_MAPPING_KEY,
114 FONT_FAMILY,
115 ICONS,
116 MANUAL_SAMPLE_CHOICE,
117 MULTIPLEYE_BUNDLE_CHOICE,
118 MULTIPLEYE_DEFAULT_DIR,
119 ONESTOP_CHOICE,
120 ONESTOP_LEGACY_SOURCE_TOKEN,
121 ONESTOP_PUBLIC_DEFAULT_DIR,
122 ONESTOP_REGIME_CHOICES,
123 ONESTOP_REGIME_LABELS,
124 ONESTOP_REGIME_SOURCE_TOKENS,
125 POTEC_DEFAULT_DIR,
126 PUBLIC_DATASETS_CHOICE,
127 RAW_GAZE_LINK_FOR_KEY,
128 RAW_GAZE_SEEDED_FOR_KEY,
129 RAW_GAZE_SNAP_RESTORE_KEY,
130 SETUP_OVERRIDE_FOR_KEY,
131 SETUP_OVERRIDE_RESTORE_KEY,
132 SETUP_OVERRIDE_SESSION_KEYS,
133 SYNTHETIC_CHOICE,
134 TRIAL_IDENTITY_CHECK_KEY,
135 TRIAL_IDENTITY_FULL_KEY,
136 UPLOAD_CHOICE,
137 UPLOAD_FILE_TYPES,
138 WIZARD_LEAVE_KEY,
139 WIZARD_STAY_KEY,
140 WORD_LABEL_COLOR,
141 benchmark_corpora_enabled,
142 icon_html,
143 language_display,
144 multipleye_enabled,
145 plural,
146 preprocessing_enabled,
147 spoken,
148 upload_limit_mb,
149)
150from scanpath_studio.controls import (
151 _LABEL_GAP,
152 FIX_FIELD_SPECS,
153 RAW_GAZE_FIELD_SPECS,
154 WORD_FIELD_SPECS,
155 _labeled,
156 _pin,
157 _sub_caption,
158 _sub_row,
159 clear_trial_filter,
160 clear_trial_filters,
161 column_mapping_ui,
162 has_active_trial_filters,
163 read_trial_filters,
164 reassert_pending_writes,
165 unique_field_labels,
166 viz_settings_from_state,
167)
168from scanpath_studio.crash_report import guarded
169from scanpath_studio.data import (
170 FIX_OPTIONAL_FIELDS,
171 IDENTITY_SCHEMA_FIELDS,
172 STIMULUS_WORDS_FLAG,
173 TRIAL_IDENTITY_SAMPLE,
174 WORD_OPTIONAL_FIELDS,
175 ReadPlan,
176 StimulusJoin,
177 adopt_source,
178 assign_derived,
179 clear_frame_cache,
180 compute_canvas_size,
181 count_trials,
182 default_filters,
183 diagnose_filters,
184 diagnose_trial_identity,
185 empty_fixations_frame,
186 empty_words_frame,
187 filter_data,
188 filter_frame_to_keys,
189 filter_to_keys,
190 filter_trials,
191 frame_cache,
192 frame_fingerprint,
193 harmonize_frames_reporting,
194 hashable_key,
195 infer_raw_gaze_schema,
196 load_onestop_server_bundle,
197 load_sample_data,
198 load_sample_raw_gaze,
199 normalize_fixations,
200 normalize_raw_gaze,
201 normalize_words,
202 onestop_data_dir,
203 onestop_full_bundle_exists,
204 plan_table_read,
205 preprocess_fixation_stage,
206 propose_fix_schema,
207 propose_raw_gaze_schema,
208 propose_word_schema,
209 raw_gaze_in_pool,
210 read_table,
211 read_table_columns,
212 read_table_sample,
213 read_tables,
214 repair_stranded_stimulus_words,
215 reset_fingerprint_memo,
216 resolve_stimulus_image_paths,
217 stamp_source,
218 text_ids,
219 trial_identity_warning,
220 trial_keys,
221 trial_mapping_columns,
222 upload_exceeds_limit,
223 uploaded_files_total_bytes,
224 validate_fix_schema,
225 validate_raw_gaze_schema,
226 validate_word_schema,
227 vouch_for_frames,
228 zip_member_split,
229)
230from scanpath_studio.dataset_table import DATASET_COUNT_FIELDS, DatasetRow
231from scanpath_studio.datasets import (
232 POTEC_FIX_SCHEMA,
233 POTEC_WORD_SCHEMA,
234 load_multipleye_server_bundle,
235 multipleye_bundle_dir,
236)
237from scanpath_studio.debug_log import (
238 debug_enabled,
239 install_log_capture,
240 log_state_change,
241 maybe_show_debug,
242 seed_debug_mode,
243 timed,
244)
245from scanpath_studio.easter_egg import render_easter_egg
246from scanpath_studio.experimental_setup import (
247 Provenance,
248 SetupSnapshot,
249 font_pt_to_px,
250)
251from scanpath_studio.html_embed import embed_html_iframe
252from scanpath_studio.menu import (
253 close_open_popovers,
254 render_nav,
255 render_top_menu,
256 view_label,
257)
258from scanpath_studio.multipart import SCREEN_ID, extract_part, part_catalog
259from scanpath_studio.persistence import (
260 PERSIST_ENV_VAR,
261 cache_failure,
262 cache_status,
263 clear_local_state,
264 clear_saved_work,
265 consume_restore_skipped,
266 discard_failed_dataset,
267 discard_failed_metadata,
268 failed_datasets,
269 failed_metadata,
270 human_size,
271 is_loopback_url,
272 local_state_restored,
273 persistence_enabled,
274 persistence_paused,
275 restore_local_state,
276 restored_from_cache,
277 restored_summary,
278 retry_cache_restore,
279 retry_failed_datasets,
280 retry_failed_metadata,
281 save_local_state,
282 server_bound_to_loopback,
283)
284from scanpath_studio.session_keys import (
285 COLUMN_MAPPING_PREFIX,
286 PARAM_CORPUS,
287 PARAM_DATASET,
288)
289from scanpath_studio.styles import get_app_css, widen_menu
290from scanpath_studio.tabs import (
291 _EDITOR_KEY_NOISE,
292 _REMAP_DIRTY_KEY,
293 _TABLE_LABELS,
294 EDITOR_NAME_FIELD_KEY,
295 EDITOR_PENDING_NAME_KEY,
296 STIMULUS_JOIN_NOTICE_KEY,
297 _build_figure_settings,
298 _render_column_mapping_section,
299 commit_builtin_setup,
300 data_scope_text,
301 dataset_editor_is_dirty,
302 discard_editor_widgets,
303 pool_filter_frames,
304 render_analysis_pool_bar,
305 render_corpus_analysis_tab,
306 render_data_health,
307 render_data_inspection_tab,
308 render_dataset_capabilities,
309 render_dataset_editor_footer,
310 render_participant_metadata_section,
311 render_settings_file,
312 render_single_trial_tab,
313 render_text_metadata_section,
314 render_trial_identity_section,
315 render_trial_metadata_section,
316)
317from scanpath_studio.tour import (
318 build_tutorial_context,
319 maybe_show_faq,
320 maybe_show_tutorial_library,
321 maybe_show_welcome_tour,
322 render_spotlight_tour,
323 render_use_case_tutorial,
324 stash_tutorial_context,
325)
326from scanpath_studio.truncation_tooltip import render_truncation_tooltips
327from scanpath_studio.url_state import (
328 CORPUS_SOURCE_TOKEN,
329 _apply_pending_trial_selection,
330 _apply_uploaded_plot_config,
331 _apply_url_preset,
332 _apply_url_trial_selection,
333 _build_share_query, # noqa: F401 re-exported for tests
334 _go_data,
335 _render_share_body,
336 apply_pending_preprocessing,
337 corpus_choice_for_slug,
338 link_dataset_notice,
339 link_sets,
340 link_setup_keys_for,
341 resolve_link_dataset,
342 scope_link_setup,
343)
345# NOTE: ``scanpath_studio.wizard`` is imported lazily inside the two functions
346# that use it (resolve_data_source, main), not here. wizard does
347# ``from . import app`` at module top, so a top-level import here forms a cycle:
348# under ``streamlit run app.py`` the script isn't registered as
349# ``scanpath_studio.app``, so wizard's ``from . import app`` re-imports app fresh,
350# re-entering this import while wizard is still half-loaded → ImportError.
351# Deferring it lets app finish loading before wizard is ever imported.
352from scanpath_studio.utils import (
353 build_combo_options,
354 combo_source,
355 extract_trial,
356 trial_id_layout,
357)
359# Re-exported under a private alias so tests can import it from `app`; keep the
360# F401 silence (it's not used by app.py itself).
361from scanpath_studio.utils import ( # noqa: F401
362 build_comparison_options as _build_comparison_options,
363)
365if TYPE_CHECKING: # pragma: no cover - typing only
366 from streamlit.delta_generator import DeltaGenerator
369def __getattr__(name):
370 """Lazily re-export the wizard helpers from ``app`` for back-compat.
372 ``scanpath_studio.wizard`` can't be imported at module load (it imports
373 ``app`` back, forming a cycle — see the note above the utils import), but
374 ``from scanpath_studio.app import _render_data_setup`` (and the other wizard
375 helpers) was a supported entry point used by tests. Resolving it here, on
376 attribute access, defers the wizard import until app is fully loaded."""
377 if name in ("_enter_add_data_wizard", "_remove_dataset", "_render_data_setup"):
378 from scanpath_studio import wizard
380 return getattr(wizard, name)
381 raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
384def public_datasets_enabled() -> bool:
385 """Whether the "Public datasets" source (PoTeC, MultiplEYE) is offered.
387 Enabled by default; set ``SCANPATH_PUBLIC_DATASETS=0`` (or ``false`` / ``no``)
388 to hide it. Read at call time so tests can toggle the env var."""
389 raw = os.environ.get("SCANPATH_PUBLIC_DATASETS", "").strip().lower()
390 return raw not in ("0", "false", "no")
393#: UX-109 follow-up. The CSS `direction: ltr` rule in `get_app_css` does not
394#: reach this: Streamlit's own widgets (the trial-position slider among them)
395#: are built on react-aria, whose `useLocale()` decides RTL straight from
396#: `navigator.language` — never from the page's CSS `direction` or a
397#: `dir="rtl"` attribute — confirmed by reading Streamlit's own shipped
398#: `Slider.*.js`, which flips its thumb's percentage (`B = 1 - B`) whenever
399#: that locale reads as RTL. A Hebrew-locale browser therefore renders the
400#: thumb at the *mirrored* position while the surrounding markup (this app's
401#: own CSS, the filled-track gradient) stays physically left-to-right,
402#: producing exactly the reported screenshot: thumb correct for a mirrored
403#: scale, fill correct for a normal one, agreeing nowhere except the ends.
404#: This app has no RTL content of its own to get right, so the fix is to make
405#: every react-aria consumer see a LTR locale regardless of the browser's:
406#: override `navigator.language`/`languages` on the *parent* window (this
407#: embed is same-origin, so it is the same `Navigator` object React reads,
408#: not a copy) and fire `languagechange`, which is the event react-aria's own
409#: locale cache listens for to invalidate itself and re-render every mounted
410#: consumer — needed because Streamlit's own chrome (top nav, main menu) can
411#: easily have already read+cached the browser's real locale before this
412#: script ever gets a chance to run.
413_FORCE_LTR_LOCALE_SCRIPT = """
414<script>
415(function () {
416 try {
417 var nav = window.parent.navigator;
418 Object.defineProperty(nav, "language", {
419 get: function () { return "en-US"; },
420 configurable: true,
421 });
422 Object.defineProperty(nav, "languages", {
423 get: function () { return ["en-US"]; },
424 configurable: true,
425 });
426 window.parent.dispatchEvent(new Event("languagechange"));
427 } catch (e) {
428 /* A browser that refuses the redefinition leaves the page as it
429 found it — no worse than before this ran. */
430 }
431})();
432</script>
433"""
436#: BUG-86. Streamlit's `help=` tooltip keeps its panel open for as long as
437#: focus is inside the trigger, and it can leave several panels in the page at
438#: once — some still open, some stuck half-closed (`data-exiting`) with no
439#: owner at all; a clicked ▶ left "Next trial." behind through reruns. CSS alone
440#: can only ask "is *some* trigger hovered?", so hovering any tooltip button
441#: brought every stale panel back. This marks a panel *owned* while its **own**
442#: trigger — the element whose `aria-describedby` names the panel's id — is
443#: `:hover` or holds `:focus-visible`, and `styles.get_app_css` hides every
444#: panel that is not, once this is running (the `data-sps-tooltip-owners` flag
445#: on `<html>`). It reads only the browser's own hover/focus state and never
446#: touches React's, so it cannot fight the component.
447#:
448#: A panel is judged the moment it appears — synchronously in a
449#: `MutationObserver`, which runs before the browser paints — so a stale one
450#: is never shown for a frame; pointer and focus moves re-judge at most once a
451#: frame. The code runs in the *parent* page's realm (a `<script>` added to its
452#: head, once per page load), not as closures from this iframe's: an iframe's
453#: listeners die with it, which is what BUG-51 had to hand-roll a heartbeat for.
454_TOOLTIP_OWNER_SCRIPT = """
455<script>
456(function () {
457 function install() {
458 var OWNED = "data-sps-tooltip-owned";
459 var PANEL = '[data-testid="stTooltipContent"], '
460 + '[data-testid="stTooltipErrorContent"]';
461 var pending = false;
462 function ownerIsActive(tip) {
463 if (!tip.id) { return false; }
464 var owner = document.querySelector(
465 '[aria-describedby~="' + CSS.escape(tip.id) + '"]'
466 );
467 return !!owner && (
468 owner.matches(":hover")
469 || owner.matches(":focus-visible")
470 || !!owner.querySelector(":focus-visible")
471 );
472 }
473 function update() {
474 pending = false;
475 var tips = document.querySelectorAll('[role="tooltip"]');
476 for (var i = 0; i < tips.length; i++) {
477 var tip = tips[i];
478 if (!tip.querySelector(PANEL)) { continue; }
479 var owned = ownerIsActive(tip);
480 if (owned !== tip.hasAttribute(OWNED)) {
481 tip.toggleAttribute(OWNED, owned);
482 }
483 }
484 }
485 function schedule() {
486 if (pending) { return; }
487 pending = true;
488 requestAnimationFrame(update);
489 }
490 ["pointerover", "pointerout", "focusin", "focusout", "keydown"].forEach(
491 function (type) { document.addEventListener(type, schedule, true); }
492 );
493 new MutationObserver(update).observe(document.body, {
494 childList: true,
495 subtree: true,
496 attributes: true,
497 attributeFilter: ["aria-describedby"],
498 });
499 update();
500 document.documentElement.setAttribute("data-sps-tooltip-owners", "");
501 }
502 try {
503 var host = window.parent;
504 if (host.__spsTooltipOwnerInstalled) { return; }
505 var script = host.document.createElement("script");
506 script.textContent = "(" + install.toString() + ")();";
507 host.document.head.appendChild(script);
508 host.__spsTooltipOwnerInstalled = true;
509 } catch (e) {
510 /* The CSS floor in styles.get_app_css still hides every panel while
511 no trigger is hovered or keyboard-focused. */
512 }
513})();
514</script>
515"""
517#: UX-193: a click on an embedded iframe — the scanpath plot above all, which
518#: spans the whole main column — never closed an open popover. Streamlit
519#: dismisses one on a `click` on the *page's* document, and a click inside an
520#: iframe lands in that iframe's document instead. What the page does see is
521#: its window losing focus to the frame, so on `blur` with an `<iframe>` now
522#: active (and not one inside a popover) this clicks every expanded popover
523#: trigger, which toggles it shut — the same move as `menu.close_open_popovers`.
524#: Installed in the parent's realm once per page load, like the script above.
525_IFRAME_CLICK_CLOSES_POPOVER_SCRIPT = """
526<script>
527(function () {
528 function install() {
529 window.addEventListener("blur", function () {
530 setTimeout(function () {
531 var active = document.activeElement;
532 if (!active || active.tagName !== "IFRAME") { return; }
533 if (active.closest('[data-st-overlay-root="true"]')) { return; }
534 document.querySelectorAll(
535 '[data-testid="stPopover"] button[aria-expanded="true"]'
536 ).forEach(function (trigger) { trigger.click(); });
537 }, 0);
538 });
539 }
540 try {
541 var host = window.parent;
542 if (host.__spsIframePopoverCloseInstalled) { return; }
543 var script = host.document.createElement("script");
544 script.textContent = "(" + install.toString() + ")();";
545 host.document.head.appendChild(script);
546 host.__spsIframePopoverCloseInstalled = true;
547 } catch (e) {
548 /* Not same-origin: clicks outside the iframes still close popovers. */
549 }
550})();
551</script>
552"""
555#: #374 F19: what a screen reader announces. Streamlit sets a widget's
556#: ``aria-label`` to its label string as written, so a switch labelled
557#: ``:material/movie: Animate`` was announced with the shortcode, and the
558#: ligature of every icon it draws (``restart_alt``, ``filter_alt``) was read
559#: as part of the button or tab around it. This strips shortcodes and ``**``
560#: from every ``aria-label`` and hides icon glyphs from assistive technology —
561#: unless the glyph is all that names its control. Labels are fixed at the
562#: source where they can be (`fields.accessible_name`); this covers the visible
563#: labels that keep an icon (toggles and checkboxes take no ``icon=``) and the
564#: glyphs Streamlit draws itself. Installed in the parent's realm once per page
565#: load, like the scripts above.
566_A11Y_NAMES_SCRIPT = """
567<script>
568(function () {
569 function install() {
570 var SHORTCODE = /:material\\/[a-z0-9_]+:/g;
571 var GLYPH = '[data-testid="stIconMaterial"], span[role="img"][translate="no"]';
572 var NAMED = 'button, a, [role="tab"], [role="button"], label, [role="option"]';
573 var pending = false;
574 function clean(value) {
575 return value.replace(SHORTCODE, " ").replace(/\\*\\*/g, "")
576 .replace(/\\s+/g, " ").trim();
577 }
578 function textWithout(el) {
579 var copy = el.cloneNode(true);
580 copy.querySelectorAll(GLYPH).forEach(function (g) { g.remove(); });
581 return (copy.textContent || "").trim();
582 }
583 function update() {
584 pending = false;
585 document.querySelectorAll('[aria-label*=":material/"], [aria-label*="**"]')
586 .forEach(function (el) {
587 var value = el.getAttribute("aria-label");
588 var fixed = clean(value);
589 if (fixed !== value) { el.setAttribute("aria-label", fixed); }
590 });
591 document.querySelectorAll(GLYPH).forEach(function (g) {
592 if (g.getAttribute("aria-hidden") === "true") { return; }
593 var owner = g.closest(NAMED);
594 if (owner && !owner.getAttribute("aria-label")
595 && !textWithout(owner)) { return; }
596 g.setAttribute("aria-hidden", "true");
597 });
598 }
599 function schedule() {
600 if (pending) { return; }
601 pending = true;
602 requestAnimationFrame(update);
603 }
604 new MutationObserver(schedule).observe(document.body, {
605 childList: true,
606 subtree: true,
607 attributes: true,
608 attributeFilter: ["aria-label"],
609 });
610 update();
611 }
612 try {
613 var host = window.parent;
614 if (host.__spsA11yNamesInstalled) { return; }
615 var script = host.document.createElement("script");
616 script.textContent = "(" + install.toString() + ")();";
617 host.document.head.appendChild(script);
618 host.__spsA11yNamesInstalled = true;
619 } catch (e) {
620 /* Not same-origin: the labels fixed at the source still read right. */
621 }
622})();
623</script>
624"""
627def configure_page() -> None:
628 """Streamlit page config + custom CSS.
630 No ``initial_sidebar_state``: nothing writes to ``st.sidebar`` any more (the
631 former sidebar groups are popovers on the top menu bar — see
632 :mod:`scanpath_studio.menu`), so Streamlit renders no sidebar chrome to
633 collapse. Embeds and welcome-tour sessions used to ask for it explicitly;
634 both now get a page with no sidebar at all, which is what they wanted.
635 """
636 st.set_page_config(
637 page_title="Scanpath Studio - Visualization of Eye Movements in Reading",
638 # UX-138 left the favicon an emoji on purpose: Streamlit draws an emoji
639 # page icon as an inline SVG, but turns a `:material/…:` one into a
640 # fonts.gstatic.com URL — a request to Google on every page load, and no
641 # icon at all offline or in the desktop bundle.
642 page_icon="👀",
643 layout="wide",
644 )
645 st.markdown(get_app_css(), unsafe_allow_html=True)
646 embed_html_iframe(_FORCE_LTR_LOCALE_SCRIPT, height=0)
647 embed_html_iframe(_TOOLTIP_OWNER_SCRIPT, height=0)
648 embed_html_iframe(_IFRAME_CLICK_CLOSES_POPOVER_SCRIPT, height=0)
649 embed_html_iframe(_A11Y_NAMES_SCRIPT, height=0)
652#: The app's wordmark, shown in Streamlit's own header (UX-62). Inside the
653#: package so it ships with a pip install — see `pyproject.toml`'s
654#: `package-data`, and `desktop/scanpath_studio.spec` for the frozen build.
655LOGO_PATH = Path(__file__).parent / "assets" / "scanpath_studio_title_logo.png"
656#: The same wordmark with light text, for the dark theme (#374 F37). Both are
657#: transparent, so neither theme draws a tile behind them.
658LOGO_DARK_PATH = LOGO_PATH.with_name("scanpath_studio_title_logo_dark.png")
661@functools.cache
662def _logo_theme_css() -> str:
663 """CSS that swaps the header wordmark for its dark variant in dark mode.
665 ``st.logo`` takes one image, and the server cannot tell the theme reliably
666 (``st.context.theme`` is wrong on first load and right after a switch). So
667 the swap happens in CSS: ``light-dark()`` follows the theme Streamlit puts
668 on the page, instantly; a browser without image support in ``light-dark()``
669 drops that line and falls back to the OS preference.
670 """
671 if not (LOGO_PATH.is_file() and LOGO_DARK_PATH.is_file()):
672 return ""
674 def _uri(path: Path) -> str:
675 return (
676 "url(data:image/png;base64,"
677 + base64.b64encode(path.read_bytes()).decode()
678 + ")"
679 )
681 light, dark = _uri(LOGO_PATH), _uri(LOGO_DARK_PATH)
682 sel = 'img[data-testid="stHeaderLogo"]'
683 return (
684 "<style>"
685 f"@media (prefers-color-scheme: dark) {{ {sel} {{ content: {dark}; }} }}"
686 f"{sel} {{ content: light-dark({light}, {dark}); }}"
687 "</style>"
688 )
691def render_app_logo() -> None:
692 """Put the wordmark in the top-left of Streamlit's header (UX-62).
694 ``st.logo`` is the only way into that strip: the nav is drawn there by
695 Streamlit itself (``st.navigation(position="top")``), and the page body —
696 where the title used to live, a row below — cannot reach up into it.
698 Must run **before** the nav, and is cheap enough to run on every rerun.
699 Falls back silently to nothing when the file is missing: a wordmark is
700 chrome, and an editable checkout that has not been reinstalled should still
701 open rather than raise on a decoration.
702 """
703 if not LOGO_PATH.is_file():
704 logging.getLogger(__name__).warning(
705 "App logo not found at %s; header left bare.", LOGO_PATH
706 )
707 return
708 st.logo(str(LOGO_PATH), size="large", link=CITATION["docs_url"])
709 if css := _logo_theme_css():
710 # A style-only `st.html` goes to Streamlit's event container: no block.
711 st.html(css)
714def _render_about_panel(host=None) -> None:
715 """The page heading — now only what the header cannot carry.
717 UX-62 moved the title into Streamlit's header as the wordmark
718 (:func:`render_app_logo`), so this no longer prints "Scanpath Studio" or its
719 one-line description; both would then appear twice, a row apart. The
720 description survives in **About** (a dialog off the ❓ Help menu) and in the
721 README.
723 ``host`` is ``menu.TopMenu.title`` — the left side of the row the settings
724 triggers share. The container is still created, and deliberately: it is
725 where anything page-level would go (see ``menu.render_top_menu``).
726 """
727 (host if host is not None else st).container(key="about_header")
730# Base URL for the DiLi Lab (UZH) people pages — three co-author links hang off
731# it, so it's factored out rather than repeated in the About markdown.
732_DILI = "https://www.cl.uzh.ch/en/research-groups/digital-linguistics/people"
735# Human labels for the trial-filter groups, used by the UX-7 empty-state report.
736_FILTER_GROUP_LABELS = {
737 "participants": "Participant",
738 "favorites": "★ Favorites only",
739 "required_tags": "With any of these tags",
740 "excluded_tags": "Excluding tags",
741}
744def _filter_diagnosis_steps(trial_filters: dict) -> list:
745 """``(label, apply)`` pairs for :func:`data.diagnose_filters` — one per
746 *active* trial filter, each applying only itself (UX-7).
748 Condition filters get one step each (named by the column) rather than being
749 lumped together, since "which of my six narrowings emptied this?" is exactly
750 the question the blanket warning used to leave unanswered.
751 """
752 steps: list = []
753 if trial_filters.get("participants") is not None:
754 chosen = trial_filters["participants"]
755 # DATA-20: a participant-grain metadata constraint resolves into this
756 # same slot, so the step has to name *and clear* whichever widgets
757 # actually produced the narrowing — otherwise the report blamed
758 # "Participant" and its Clear button popped `filter_participants`, a
759 # no-op against a `filter_meta_*` selection.
760 meta_keys = tuple(trial_filters.get("participant_filter_keys") or ())
761 label = f"{_FILTER_GROUP_LABELS['participants']} ({len(chosen)} selected)"
762 if meta_keys:
763 label = (
764 "By participant"
765 if not st.session_state.get("filter_participants")
766 else f"{label} + by participant"
767 )
768 steps.append(
769 (
770 label,
771 lambda w, f, p=chosen: filter_trials(w, f, participants=p),
772 ("filter_participants", *meta_keys),
773 )
774 )
775 keys_by_col = trial_filters.get("metadata_keys") or {}
776 # DATA-66: each filter by the dataset's own name for its column, as the
777 # filter panel titles it; UX-149: no two share a label.
778 names = unique_field_labels(
779 [
780 *(trial_filters.get("metadata") or {}),
781 *(trial_filters.get("ranges") or {}),
782 ],
783 active_all(st.session_state).label,
784 )
785 for col, allowed in (trial_filters.get("metadata") or {}).items():
786 label = f"{names[col]} = {', '.join(sorted(map(str, allowed))[:4])}"
787 if len(allowed) > 4:
788 label += ", …"
789 steps.append(
790 (
791 label,
792 lambda w, f, c=col, a=allowed: filter_trials(w, f, metadata={c: a}),
793 (keys_by_col.get(col, f"filter_{col}"),),
794 )
795 )
796 # UX-49: a range narrows too, so it is one of the things that can empty the
797 # pool and has to be named in the diagnosis alongside the categorical ones.
798 dropping = set(trial_filters.get("ranges_drop_unknown") or ())
799 for col, bounds in (trial_filters.get("ranges") or {}).items():
800 label = f"{names[col]} between {bounds[0]:,.10g} and {bounds[1]:,.10g}"
801 if col in dropping:
802 label += " (unknown values excluded)"
803 steps.append(
804 (
805 label,
806 lambda w, f, c=col, b=bounds, d=col in dropping: filter_trials(
807 w, f, ranges={c: b}, drop_unknown=(c,) if d else None
808 ),
809 (keys_by_col.get(col, f"filter_{col}_range"),),
810 )
811 )
813 def _annotation_step(name: str, keys: tuple, **kwargs):
814 def _apply(w, f):
815 frame = w if f.empty else f
816 if frame.empty:
817 return w, f
818 present = {
819 (str(p), str(t))
820 for p, t in zip(frame["participant_id"], frame["trial_id"])
821 }
822 return filter_to_keys(w, f, set(filter_keys(list(present), **kwargs)))
824 steps.append((name, _apply, keys))
826 if trial_filters.get("favorites_only"):
827 _annotation_step(
828 _FILTER_GROUP_LABELS["favorites"],
829 ("filter_favorites",),
830 favorites_only=True,
831 )
832 if trial_filters.get("required_tags"):
833 tags = trial_filters["required_tags"]
834 _annotation_step(
835 f"{_FILTER_GROUP_LABELS['required_tags']}: {', '.join(tags)}",
836 ("filter_req_tags",),
837 required_tags=tags,
838 )
839 if trial_filters.get("excluded_tags"):
840 tags = trial_filters["excluded_tags"]
841 _annotation_step(
842 f"{_FILTER_GROUP_LABELS['excluded_tags']}: {', '.join(tags)}",
843 ("filter_exc_tags",),
844 excluded_tags=tags,
845 )
846 return steps
849def _render_empty_after_filtering(
850 words_all: pd.DataFrame,
851 fixations_all: pd.DataFrame,
852 trial_filters: dict,
853 filter_frames: tuple,
854) -> None:
855 """UX-7(a): say *which* filter emptied the pool, and offer a way out.
857 The old message was one blanket "No data after filtering" for every cause,
858 which left the user to bisect their own filters by hand. This measures each
859 active filter against the unfiltered dataset, names the one(s) that leave
860 nothing on their own (or reports the combination when each is individually
861 fine), and offers to clear **that filter alone** as well as all of them.
863 Rendered as one bordered panel: the previous version stacked an `st.warning`
864 banner, a markdown list and a button as three visually unrelated blocks, so
865 the diagnosis didn't read as belonging to the message above it.
866 """
867 total = count_trials(words_all, fixations_all)
868 if not has_active_trial_filters():
869 # Nothing is filtering, so the dataset itself is empty — a different
870 # problem, and telling the user to loosen filters would be a wild goose
871 # chase.
872 with st.container(border=True, key="empty_state_panel"):
873 st.markdown("#### This dataset has no trials to show")
874 st.markdown(
875 "Open another dataset, or check this one's column mapping "
876 f"under {ICONS['view_data']} **Data Management → Edit dataset**."
877 )
878 return
880 report = diagnose_filters(
881 words_all, fixations_all, _filter_diagnosis_steps(trial_filters)
882 )
883 culprits = [row for row in report if row["empties"]]
884 with st.container(border=True, key="empty_state_panel"):
885 st.markdown(
886 f"#### No trials match your filters\n"
887 f"**0** of the **{plural(total, 'trial')}** in this dataset get through."
888 )
889 rows = culprits or report
890 if culprits:
891 st.markdown(
892 "On its own, this leaves nothing:"
893 if len(culprits) == 1
894 else "Each of these leaves nothing on its own:"
895 )
896 elif report:
897 st.markdown(
898 "Every filter keeps something on its own — it's the "
899 "**combination** that leaves nothing:"
900 )
901 for i, row in enumerate(rows):
902 text_col, clear_col = st.columns([5, 1], vertical_alignment="center")
903 kept = "" if row["empties"] else f" — keeps {row['kept']:,} of {total:,}"
904 text_col.markdown(f"{row['label']}{kept}")
905 if row["keys"]:
906 clear_col.button(
907 "Clear",
908 key=f"clear_one_filter_{i}",
909 on_click=clear_trial_filter,
910 args=tuple(row["keys"]),
911 # BUG-115: the funnel is not drawn beside this panel, so the
912 # other filters are re-derived from the frames it filters.
913 kwargs={"frames": filter_frames},
914 help=f"Reset only this filter — {row['label']}.",
915 width="stretch",
916 )
917 st.button(
918 "✕ Clear all filters",
919 key="clear_all_trial_filters",
920 type="primary",
921 on_click=clear_trial_filters,
922 help="Reset every trial filter.",
923 )
926#: UX-136 — the kinds `persistence.restored_summary` counts, in the order the
927#: *Saved on this computer* section lists them, with their singular/plural nouns.
928#: DATA-38 added the attached metadata tables.
929_RESTORED_KIND_NOUNS = (
930 ("datasets", "dataset", "datasets"),
931 ("annotations", "annotation", "annotations"),
932 ("designs", "design", "designs"),
933 ("metadata", "metadata table", "metadata tables"),
934)
937def _restored_recap(session=None) -> str:
938 """What came back from the recovery cache, as a phrase for the toast.
940 "2 datasets and 3 annotations" — only the kinds that actually restored, so
941 the sentence never pads itself with the zeros the panel legitimately shows.
942 Falls back to "your last session" if it is somehow called with nothing to
943 report, which keeps the toast a sentence rather than a hole.
944 """
945 summary = restored_summary(st.session_state if session is None else session)
946 parts = [
947 f"{summary[key]:,} {singular if summary[key] == 1 else plural}"
948 for key, singular, plural in _RESTORED_KIND_NOUNS
949 if summary.get(key)
950 ]
951 if not parts:
952 return "your last session"
953 if len(parts) == 1:
954 return parts[0]
955 return f"{', '.join(parts[:-1])} and {parts[-1]}"
958def _pick_download_folder() -> None:
959 """📁 beside the Download folder box: a native picker, applied next run."""
960 chosen = _pick_directory_dialog()
961 if chosen:
962 st.session_state[f"{DOWNLOAD_DIR_KEY}_picked"] = chosen
963 else:
964 st.session_state[f"{DOWNLOAD_DIR_KEY}_no_picker"] = True
967def _render_download_folder_section(host) -> None:
968 """🗂️ Data → **Download folder** (UX-184).
970 One folder for every ⬇ Download, so a user chooses where the corpora go once
971 rather than per dataset — the per-dataset Data directory box still
972 overrides it. Before this the folder was implicit (the checkout's ``data/``,
973 or the per-user data home: ``%LOCALAPPDATA%`` on Windows) and the page only
974 ever said ``data/PoTeC``. Not drawn where the app may not touch local
975 folders (S2) — there the server's configuration decides.
976 """
977 if not local_filesystem_enabled():
978 return
979 picked = st.session_state.pop(f"{DOWNLOAD_DIR_KEY}_picked", None)
980 if picked:
981 st.session_state[DOWNLOAD_DIR_KEY] = picked
982 st.session_state.setdefault(DOWNLOAD_DIR_KEY, "")
983 host.divider()
984 host.subheader(f"{ICONS['download']} Download folder")
985 host.caption(
986 "Where **Download** saves a public dataset, each in its own subfolder. "
987 "Leave it blank for the default. A dataset's own *Data directory* box "
988 "overrides it."
989 )
990 text_col, browse_col = host.columns([4, 1])
991 text_col.text_input(
992 "Download folder",
993 key=DOWNLOAD_DIR_KEY,
994 placeholder=str(_default_download_folder()),
995 label_visibility="collapsed",
996 # Rendered only on the Data page's overview; without this the choice
997 # would be dropped the first run another view is open (BUG-15).
998 persist_state="session",
999 )
1000 browse_col.button(
1001 # UX-200: named for screen readers; the folder icon is all that shows.
1002 f"{ICONS['folder']} {spoken('Choose the download folder')}",
1003 wrap=True,
1004 key=f"{DOWNLOAD_DIR_KEY}_browse",
1005 help="Browse for a folder",
1006 on_click=_pick_download_folder,
1007 )
1008 if st.session_state.pop(f"{DOWNLOAD_DIR_KEY}_no_picker", False):
1009 host.caption("Folder picker unavailable here — type or paste the path.")
1010 host.markdown(f"**Saving to:** `{download_folder()}`")
1013def _retry_cached_datasets() -> None:
1014 """``on_click``: read the held-back cached datasets again."""
1015 retry_failed_datasets(st.session_state)
1018def _remove_cached_dataset(name: str) -> None:
1019 """``on_click``: delete one held-back dataset's entry and files from the cache."""
1020 discard_failed_dataset(st.session_state, name)
1023def _retry_cached_metadata() -> None:
1024 """``on_click``: read the held-back metadata tables again."""
1025 retry_failed_metadata(st.session_state)
1028def _remove_cached_metadata() -> None:
1029 """``on_click``: delete the held-back metadata tables' stored copy."""
1030 discard_failed_metadata(st.session_state)
1033def _retry_unreadable_cache(app_url: str) -> None:
1034 """``on_click``: try the whole cache again, as a reload would."""
1035 retry_cache_restore(st.session_state, app_url)
1038def _clear_unreadable_cache() -> None:
1039 """``on_click``: delete a cache that cannot be read; saving resumes."""
1040 clear_local_state(st.session_state)
1043def render_cache_recovery_notice(host, app_url: str, *, key: str) -> bool:
1044 """What of the recovery cache this session could not restore, with actions.
1046 Drawn where the recovery notices go (the page's notices) and again in
1047 🗂️ Data → *Saved on this computer*, with ``key`` keeping the two sets of
1048 buttons apart. Two cases, never the third — a cache that is absent, or was
1049 cleared on purpose, is not a failure and says nothing:
1051 - **a dataset** whose files are missing or unreadable. The rest restored;
1052 this one is held back, and kept in the cache — every save writes it back
1053 as it was — so **Retry** can read it once its file is back, and **Remove
1054 saved copy** deletes it.
1055 - **the cache as a whole** (a manifest that cannot be read). Saving is
1056 paused so this session cannot replace it; **Retry** reads it again and
1057 **Delete saved data** deletes it, after which saving resumes.
1059 The metadata tables' file is held back the same way as a dataset (round
1060 10): kept as it is until **Retry** reads it or **Remove saved copy**
1061 deletes that copy alone.
1063 Returns whether anything was drawn.
1064 """
1065 failure = cache_failure(st.session_state)
1066 failed = failed_datasets(st.session_state)
1067 metadata_failure = failed_metadata(st.session_state)
1068 if not failure and not failed and metadata_failure is None:
1069 return False
1070 box = host.container(border=True)
1071 if failure:
1072 box.warning(
1073 f"What was saved on this computer couldn't be read — {failure}. "
1074 "Nothing was restored, and saving is paused so it stays as it is.",
1075 icon=ICONS["warning"],
1076 )
1077 row = box.container(horizontal=True)
1078 row.button(
1079 "Retry",
1080 icon=ICONS["refresh"],
1081 key=f"{key}_retry_cache",
1082 on_click=_retry_unreadable_cache,
1083 args=(app_url,),
1084 )
1085 row.button(
1086 "Delete saved data",
1087 icon=ICONS["delete"],
1088 key=f"{key}_clear_cache",
1089 on_click=_clear_unreadable_cache,
1090 help="Permanently delete everything saved on this computer. Saving "
1091 "resumes.",
1092 )
1093 return True
1094 if metadata_failure is not None:
1095 box.warning(
1096 "The metadata tables saved on this computer couldn't be restored — "
1097 f"{metadata_failure}. They are kept as they are until "
1098 "you retry or remove them; tables you attach meanwhile are not saved.",
1099 icon=ICONS["warning"],
1100 )
1101 row = box.container(horizontal=True)
1102 row.button(
1103 "Retry",
1104 icon=ICONS["refresh"],
1105 key=f"{key}_retry_metadata",
1106 on_click=_retry_cached_metadata,
1107 )
1108 row.button(
1109 "Remove saved copy",
1110 icon=ICONS["delete"],
1111 key=f"{key}_remove_metadata",
1112 on_click=_remove_cached_metadata,
1113 help="Delete the stored metadata tables. Datasets and annotations "
1114 "are kept.",
1115 )
1116 if not failed:
1117 return True
1118 box.warning(
1119 f"{plural(len(failed), 'dataset')} saved on this computer couldn't be "
1120 "restored. Everything else came back. "
1121 f"{'They are' if len(failed) != 1 else 'It is'} kept until you retry or "
1122 f"remove {'them' if len(failed) != 1 else 'it'}.",
1123 icon=ICONS["warning"],
1124 )
1125 for index, (name, reason) in enumerate(sorted(failed.items())):
1126 row = box.container(horizontal=True, vertical_alignment="center")
1127 row.markdown(f"**{name}** — {reason}")
1128 row.button(
1129 "Retry",
1130 icon=ICONS["refresh"],
1131 key=f"{key}_retry_{index}",
1132 on_click=_retry_cached_datasets,
1133 )
1134 row.button(
1135 "Remove saved copy",
1136 icon=ICONS["delete"],
1137 key=f"{key}_remove_{index}",
1138 on_click=_remove_cached_dataset,
1139 args=(name,),
1140 help="Delete this dataset's stored copy. Its annotations are kept.",
1141 )
1142 return True
1145def _render_saved_here_section(app_url: str, host) -> None:
1146 """🗂️ Data → **Saved on this computer** (UX-179; ENG-30 underneath).
1148 The foot of the Data page's overview, and a read-out only: what the
1149 recovery cache holds and the folder it is in. It was the retired 💾 Session
1150 dialog's first block; it lives here because its count is "datasets **you
1151 added**", which is the table above it.
1153 UX-179 left it no controls: opting out is a launch choice
1154 (``run --no-persist`` / ``SCANPATH_STUDIO_PERSIST=0``, in the FAQ). #374 F33
1155 gave it one back, **Clear what is saved…**, behind a confirmation listing
1156 what goes (``persistence.clear_saved_work``) — the CLI's
1157 ``scanpath-studio cache --clear`` was the only way before. BUG-71's pause,
1158 after a restore that crashed, is the other in-session state it names.
1160 Drawn *after* this run's ``save_local_state`` (``main``'s
1161 ``_finish_page``), so the status line reports the write that just happened.
1162 ``cache_status`` re-reads the manifest each run — a few ``stat`` calls and a
1163 small JSON — rather than being cached: a status line that lags what it
1164 reports is worse than none.
1165 """
1166 status = cache_status(url=app_url)
1167 host.divider()
1168 host.subheader(f"{ICONS['recovery']} Saved on this computer")
1169 if not status["enabled"]:
1170 turned_off = status["override"] == "off"
1171 host.caption(
1172 (
1173 "**Turned off.** Scanpath Studio was started with saving "
1174 "switched off, so it keeps your work in memory only"
1175 if turned_off
1176 else "**Not available here.** This deployment keeps your work "
1177 "in memory only"
1178 )
1179 + " — closing or refreshing the tab loses the datasets you "
1180 "uploaded, their column mappings, your annotations and designs. Export the "
1181 "annotations from **Annotations** above, and the figure's settings "
1182 f"from {ICONS['view_scanpath']} Scanpath → {ICONS['share']} Share → **File**."
1183 )
1184 if turned_off:
1185 # The switch itself, for whoever launches the app — kept out of the
1186 # sentence above, which is for everyone.
1187 host.caption(
1188 f"Started with `--no-persist` or `{PERSIST_ENV_VAR}=0`; start "
1189 "it without either to save again."
1190 )
1191 return
1193 host.caption(
1194 "Saved as you work and reopened next time — your added datasets, "
1195 "annotations, designs, and the view, trial and plot settings you left. "
1196 "Nothing is uploaded anywhere."
1197 )
1198 if restored_from_cache(st.session_state):
1199 host.success("Recovered when the app opened.", icon=ICONS["recovery"])
1200 if status["exists"] and status["readable"]:
1201 n_sets = len(status["datasets"])
1202 # "datasets **you added**", not "datasets": only an upload is copied
1203 # here — the bundled demo and the public corpora reload from their own
1204 # source — so the count reads 0 while one of those is open, which looked
1205 # like a bug until the line said which datasets it was counting.
1206 host.markdown(
1207 f"**Saved here:** {n_sets} dataset{'s' if n_sets != 1 else ''} "
1208 f"you added · {status['annotations']} annotation"
1209 f"{'s' if status['annotations'] != 1 else ''} · "
1210 f"{status['designs']} design"
1211 f"{'s' if status['designs'] != 1 else ''} · "
1212 # DATA-38 — named only when there are any, so the common line keeps
1213 # its length.
1214 + (
1215 f"{status['metadata']} metadata table"
1216 f"{'s' if status['metadata'] != 1 else ''} · "
1217 if status.get("metadata")
1218 else ""
1219 )
1220 + f"{human_size(status['bytes'])}"
1221 )
1222 elif status["exists"] and not cache_failure(st.session_state):
1223 host.warning(
1224 "What was saved here can't be read (written by a different version, "
1225 "or incomplete).",
1226 icon=ICONS["warning"],
1227 )
1228 elif not status["exists"]:
1229 host.caption("Nothing saved yet — the first change you make is saved here.")
1230 render_cache_recovery_notice(host, app_url, key="saved_here_recovery")
1231 host.markdown(f"**Folder:** `{status['directory']}`")
1232 # #374 F33. The unreadable-cache box keeps its own Clear button.
1233 if status["exists"] and not cache_failure(st.session_state):
1234 host.button(
1235 "Clear what is saved…",
1236 icon=ICONS["delete"],
1237 key="saved_here_clear",
1238 help="Delete everything listed above from this computer and start "
1239 "over, after a confirmation.",
1240 on_click=_arm_clear_saved,
1241 )
1242 if st.session_state.pop(CLEAR_SAVED_REQUEST_KEY, False):
1243 _clear_saved_dialog(app_url)
1244 if persistence_paused(st.session_state) and not cache_failure(st.session_state):
1245 # BUG-71 — the only pause left: the last launch never finished opening
1246 # with this cache, so this session neither restored nor overwrites it.
1247 host.caption(
1248 "Saving is paused for this session, so the copy above stays as it "
1249 "was. Reload to try restoring it again, or delete it with "
1250 "`scanpath-studio cache --clear`."
1251 )
1254#: #374 F33 — *Clear what is saved…*'s request flag, served right under the
1255#: button (the section is the last thing a run draws, so nothing waits on it).
1256CLEAR_SAVED_REQUEST_KEY = "_clear_saved_requested"
1257#: Set by :func:`_clear_and_start_over` on the emptied session, so the first run
1258#: of the fresh start says what happened.
1259STARTED_OVER_KEY = "_started_over"
1260#: What survives starting over: having seen or dismissed the welcome tour is not
1261#: "your work", and the tour reopening over a fresh start would be noise.
1262_KEPT_ON_START_OVER = ("_tour_dismissed", "tour_seen")
1265def _clear_and_start_over() -> None:
1266 """*Clear what is saved…* → Delete: the files, then the session (#374 F33).
1268 A callback, so it runs before anything else in the rerun. The link's
1269 parameters go too, or they would seed the fresh session again."""
1270 kept = {
1271 k: st.session_state[k] for k in _KEPT_ON_START_OVER if k in st.session_state
1272 }
1273 clear_saved_work(st.session_state)
1274 st.session_state.update(kept)
1275 st.session_state[STARTED_OVER_KEY] = True
1276 st.query_params.clear()
1279def announce_start_over() -> None:
1280 """Say, once, that the app started over after *Clear what is saved…*."""
1281 if st.session_state.pop(STARTED_OVER_KEY, False):
1282 st.toast(
1283 "Cleared what was saved on this computer. The app started over.",
1284 icon=ICONS["recovery"],
1285 )
1288def _arm_clear_saved() -> None:
1289 st.session_state[CLEAR_SAVED_REQUEST_KEY] = True
1292def saved_items(status: dict) -> list[str]:
1293 """What *Clear what is saved…* deletes, one line each (#374 F33)."""
1294 names = [str(entry["name"]) for entry in status.get("datasets") or []]
1295 names += [str(entry["name"]) for entry in status.get("damaged") or []]
1296 items = []
1297 if names:
1298 listed = ", ".join(f"`{name}`" for name in names)
1299 items.append(f"{plural(len(names), 'dataset')} you added: {listed}")
1300 if status.get("annotations"):
1301 items.append(plural(int(status["annotations"]), "annotation"))
1302 if status.get("designs"):
1303 items.append(f"{plural(int(status['designs']), 'saved design')}")
1304 if status.get("metadata"):
1305 items.append(plural(int(status["metadata"]), "metadata table"))
1306 items.append("the view, trial and plot settings you left")
1307 return items
1310@st.dialog(f"{ICONS['delete']} Clear what is saved?")
1311@guarded()
1312def _clear_saved_dialog(app_url: str) -> None:
1313 """Confirm *Clear what is saved…*, listing what it deletes (#374 F33)."""
1314 st.markdown(
1315 "This deletes from this computer:\n\n"
1316 + "\n".join(f"- {item}" for item in saved_items(cache_status(url=app_url)))
1317 )
1318 st.caption(
1319 "The app then starts over, as on a first visit. Your original files are "
1320 "not touched. There is no undo."
1321 )
1322 cancel, confirm = st.columns(2)
1323 if cancel.button("Cancel", key="saved_here_clear_cancel", width="stretch"):
1324 st.rerun()
1325 # A callback, so the delete happens before anything else in the rerun
1326 # (the nav may rerun the script before this dialog is reached again).
1327 confirm.button(
1328 "Delete",
1329 icon=ICONS["delete"],
1330 type="primary",
1331 key="saved_here_clear_confirm",
1332 width="stretch",
1333 on_click=_clear_and_start_over,
1334 )
1335 if st.session_state.get(STARTED_OVER_KEY):
1336 st.rerun() # the whole page, which closes this dialog
1339def _arm_about() -> None:
1340 """``on_click`` callback for the About button: request the dialog.
1342 Same shape as ``tour._arm_faq`` — a dialog can't be opened from a callback,
1343 so this only sets a flag :func:`maybe_show_about` serves early in ``main``.
1344 """
1345 st.session_state["_about_dialog_requested"] = True
1348#: #139 — the last *Check for updates* answer, shown under the button until the
1349#: dialog is opened again. A plain session key, not in the recovery cache's
1350#: allowlist, so it is never written to disk.
1351_UPDATE_CHECK_KEY = "_about_update_check"
1354def maybe_show_about() -> None:
1355 """Open the About dialog if the ❓ Help menu button armed it.
1357 Call from ``main()`` beside ``maybe_show_faq``, BEFORE the heavy data / plot
1358 work: the button renders at the very *bottom* of ``main()``, so serving the
1359 dialog from its return value would leave the modal waiting on the whole
1360 rerun (including the ~10 s plot embeds).
1361 """
1362 if st.session_state.pop("_about_dialog_requested", False):
1363 st.session_state.pop(_UPDATE_CHECK_KEY, None)
1364 _about_dialog()
1367@st.dialog(f"{ICONS['about']} About Scanpath Studio", width="large")
1368@guarded()
1369def _about_dialog() -> None:
1370 """The About modal: version, authors, links, citation, AI-assistance note."""
1371 from scanpath_studio import __release__, __version__
1373 # The button that opened this sits inside the ❓ Help popover, whose open
1374 # state is client-side — without this it floats on top of the modal.
1375 close_open_popovers()
1377 bibtex = (
1378 "@software{Shubi_Scanpath_Studio_2026,\n"
1379 "author = {Shubi, Omer and Gruteke Klein, Keren and Grossman, Maya and "
1380 "Lion, Ella and "
1381 'Jakobi, Deborah N. and Reich, David R. and J{\\"a}ger, Lena and '
1382 "Berzak, Yevgeni},\n"
1383 f"doi = {{{CITATION['doi']}}},\n"
1384 "license = {MIT},\n"
1385 "month = jun,\n"
1386 "title = {{Scanpath Studio}},\n"
1387 f"url = {{{CITATION['url']}}},\n"
1388 f"version = {{{__release__}}},\n"
1389 "year = {2026}\n"
1390 "}"
1391 )
1392 st.markdown(
1393 f"**Scanpath Studio** v{__version__} — interactive visualization of eye "
1394 "movements in reading."
1395 )
1396 _render_build_and_updates()
1397 st.markdown(
1398 f"""
1399Developed by [Omer Shubi](https://omershubi.github.io/),
1400[Keren Gruteke Klein](https://kerengruteke.github.io/),
1401[Maya Grossman](https://www.linkedin.com/in/maya-harram-32b547292/),
1402[Ella Lion](https://ella-lion.github.io/),
1403[Deborah N. Jakobi]({_DILI}/lab-members/jakobi.html),
1404[David R. Reich]({_DILI}/lab-members/reich.html),
1405[Lena Jäger]({_DILI}/group-leader/jaeger.html), and
1406[Yevgeni Berzak](https://dds.technion.ac.il/people/academic-staff/yevgeni-berzak/).
1408{ICONS["docs"]} [Documentation]({CITATION["docs_url"]}) ↗ ·
1409{ICONS["code"]} [Code]({CITATION["url"]}) ↗ ·
1410{ICONS["doi"]} [DOI](https://doi.org/{CITATION["doi"]}) ↗
1411"""
1412 )
1413 # UX-16: the BibTeX block is tall enough to push everything above it out
1414 # of view, so it opens on demand — but it stays a named section of its
1415 # own (a bold label) rather than a footnote, since "how do I cite this?" is
1416 # the single most common reason to open About. The dividers that used to
1417 # separate the three blocks are gone (the user's call): on a modal this
1418 # short the bold headings already carry the split, and three rules in half a
1419 # screen read as clutter.
1420 st.markdown(
1421 f"**{ICONS['docs']} Citing Scanpath Studio** — a paper is in preparation."
1422 )
1423 with st.expander("Show BibTeX", expanded=False):
1424 st.code(bibtex, language="bibtex", wrap_lines=True)
1425 st.markdown(
1426 """
1427If you use the bundled demo data, also cite
1428[OneStop Eye Movements](https://doi.org/10.1038/s41597-025-06272-2)
1429(Berzak et al., 2025, *Scientific Data*).
1430"""
1431 )
1432 # UX-20. A bare "built with AI, there may be bugs" is unfalsifiable, so
1433 # this points at what the reader can verify — not at how much effort went
1434 # in, which they have no way to check. Deliberately not a liability
1435 # disclaimer either: MIT already carries that. The heading says the
1436 # "built with AI assistance" half, so the prose no longer repeats it.
1437 st.markdown(f"**{ICONS['ai']} Built with AI assistance**")
1438 st.markdown(
1439 f"""
1440Cross-check results before publishing.
1442{ICONS["bug"]} Something looks wrong? [Report a bug]({CITATION["bug_report_url"]}) ↗
1443with the version above, your operating system, and how you run the app.
1445{ICONS["question"]} Questions go to [Discussions → Q&A]({CITATION["questions_url"]}) ↗,
1446and feature requests to [an issue]({CITATION["url"]}/issues) ↗.
1447"""
1448 )
1449 # #374 F31: the Debug drawer opens from here, not from the ❓ Help menu.
1450 # A full rerun closes this dialog; `maybe_show_debug` then opens the drawer.
1451 if st.button(
1452 "Debug",
1453 icon=ICONS["debug"],
1454 type="tertiary",
1455 key="about_open_debug",
1456 help="The debug log and a snapshot of what's loaded, for a bug report.",
1457 ):
1458 from scanpath_studio.debug_log import _arm_debug
1460 _arm_debug()
1461 st.rerun()
1464def _update_check_offered() -> bool:
1465 """#139: *Check for updates* only where updating means something — a local
1466 run or the desktop app, i.e. a server on loopback alone. The hosted demo
1467 runs the `stable` branch, and its visitors have nothing to update."""
1468 return server_bound_to_loopback()
1471@st.cache_data(ttl=600, show_spinner=False)
1472def _latest_release_cached() -> update_check.Release:
1473 """GitHub's latest release, kept ten minutes: a second click, or a second
1474 session on this machine, doesn't spend another of the 60 anonymous requests
1475 an hour. A failure raises, and `st.cache_data` keeps no failed result."""
1476 return update_check.latest_release()
1479def _build_info():
1480 """This process's build (a seam the About tests pin)."""
1481 from scanpath_studio.build_info import build_info
1483 return build_info()
1486#: #394: how long About goes on saying how the last update ended.
1487LAST_UPDATE_SHOWN_S = 7 * 24 * 3600
1490def _render_last_update(last: desktop_update.UpdateResult, current: str) -> None:
1491 """#385: the outcome the update helper left behind, across the restart.
1493 Said only of the build that is running: a record that names neither
1494 ``current`` version — a later install by hand — says nothing, and neither
1495 does one older than ``LAST_UPDATE_SHOWN_S``.
1496 """
1497 if last.at is not None and time.time() - last.at > LAST_UPDATE_SHOWN_S:
1498 return
1499 because = f": {last.reason}" if last.reason else ""
1500 if last.status == "updated":
1501 if last.version == current:
1502 st.caption(f"Updated from v{last.previous}.")
1503 elif last.previous == current:
1504 # This build is the one that stayed.
1505 st.warning(
1506 f"The update to v{last.version} didn't go through{because}. "
1507 f"This is still v{last.previous}.",
1508 icon=ICONS["warning"],
1509 )
1510 elif last.version == current:
1511 # A double failure kept the new version's files in place.
1512 st.warning(
1513 f"The update to v{last.version} didn't finish cleanly{because}. If "
1514 f"something misbehaves, download v{last.version} again from the "
1515 "release page.",
1516 icon=ICONS["warning"],
1517 )
1520def _render_build_and_updates() -> None:
1521 """#139: which build this is, and — on a local run — *Check for updates*."""
1522 info = _build_info()
1523 if info.version != info.release:
1524 st.caption(info.describe())
1525 # #385: how the desktop app's last update ended — None anywhere else.
1526 last = desktop_update.last_result()
1527 if last is not None:
1528 _render_last_update(last, info.version)
1529 if not _update_check_offered():
1530 return
1531 if st.button(
1532 "Check for updates",
1533 icon=ICONS["update"],
1534 key="about_check_updates",
1535 help="Asks GitHub for the latest release — the only time the app goes "
1536 "online for this.",
1537 ):
1538 # One opaque request of at most 5 s inside a dialog: a spinner, not a
1539 # UX-165 loading card.
1540 with st.spinner("Asking GitHub…"):
1541 st.session_state[_UPDATE_CHECK_KEY] = update_check.check_for_updates(
1542 latest=_latest_release_cached
1543 )
1544 result = st.session_state.get(_UPDATE_CHECK_KEY)
1545 if result is not None:
1546 _render_update_result(result)
1549def _render_update_result(result: update_check.UpdateCheck) -> None:
1550 """One *Check for updates* answer: the sentence, then what to do about it."""
1551 if result.status == "up_to_date":
1552 st.success(result.message, icon=ICONS["success"])
1553 return
1554 if result.status == "ahead":
1555 st.info(result.message, icon=ICONS["info"])
1556 return
1557 if result.status == "error":
1558 st.warning(result.message, icon=ICONS["warning"])
1559 return
1560 st.info(result.message, icon=ICONS["update"])
1561 if result.install_kind == "desktop":
1562 _render_desktop_update(result)
1563 elif result.command:
1564 st.code(result.command, language="bash")
1565 st.caption("Then restart the app.")
1566 if result.latest is not None:
1567 st.markdown(f"[What's new in v{result.latest.version}]({result.latest.url}) ↗")
1570def _render_desktop_update(result: update_check.UpdateCheck) -> None:
1571 """#385: **Update & restart** when this app can update itself, and the
1572 download either way — beside the button, or instead of it with the reason."""
1573 install = desktop_update.current_install()
1574 reason = desktop_update.refusal(result, install)
1575 clicked = False
1576 if reason is None:
1577 clicked = st.button(
1578 "Update & restart",
1579 type="primary",
1580 icon=ICONS["update"],
1581 key="about_update_restart",
1582 help="Downloads the new version, checks and tests it, then restarts "
1583 "into it. Your datasets and settings come back with it.",
1584 )
1585 else:
1586 st.caption(reason)
1587 if result.download is not None:
1588 label = (
1589 "Download instead"
1590 if reason is None
1591 else f"Download {result.download.name} ({human_size(result.download.size)})"
1592 )
1593 st.link_button(label, result.download.url, icon=ICONS["download"])
1594 if clicked:
1595 _run_desktop_update(result, install)
1598def _stop_desktop_update(task_key: tuple) -> None:
1599 """#385: Cancel on the update card — the download stops at its next chunk."""
1600 progress.cancel(task_key)
1603def _run_desktop_update(
1604 result: update_check.UpdateCheck, install: desktop_update.Install
1605) -> None:
1606 """Download, check and test under a card, then hand over to the helper and quit."""
1607 task_key = ("desktop_update", loading.session_id())
1608 try:
1609 with loading.card(
1610 st.empty(),
1611 key="desktop_update",
1612 title=f"Updating to v{result.latest.version}",
1613 steps=desktop_update.STEPS,
1614 step_list=True,
1615 task_key=task_key,
1616 cancel=loading.Cancel(
1617 "Cancel update", _stop_desktop_update, args=(task_key,)
1618 ),
1619 ):
1620 plan = desktop_update.prepare(result, install)
1621 # Its handshake (up to HANDSHAKE_TIMEOUT_S) is the card's last
1622 # step, Restarting, and Cancel still stops it there.
1623 desktop_update.start_swap(plan)
1624 # At once: the helper is waiting for this process to quit, and a
1625 # rerun (a click on Cancel, now too late) would stop this script
1626 # at its next Streamlit call, before a later exit_soon.
1627 desktop_update.exit_soon()
1628 except desktop_update.UpdateFailed as error:
1629 message = str(error)
1630 if "nothing was changed" not in message.lower():
1631 message += " Nothing was changed."
1632 st.error(message, icon=ICONS["error"])
1633 return
1634 st.success(
1635 f"Restarting into v{plan.version}. A new window opens when it's ready "
1636 "(on Windows that can take a minute or two); you can close this one.",
1637 icon=ICONS["update"],
1638 )
1641# --- Public-dataset access UI (directory + expected files + download) --------
1642# Shared by the per-corpus loaders below. Each corpus shows the on-disk layout
1643# it expects (so a user who already downloaded the data knows what to drop
1644# where) and a found-vs-missing status; downloadable corpora also get a Download
1645# button. The per-source participant/text narrowing was removed — every loader
1646# now reads the whole corpus and the global **Narrow by** trial filters scope it.
1648_POTEC_STRUCTURE_MD = """\
1649**Expected layout** — a clone of
1650[DiLi-Lab/PoTeC](https://github.com/DiLi-Lab/PoTeC) with its data files, or any
1651folder you let **Download** populate:
1652```
1653<dir>/
1654├─ eyetracking_data/
1655│ └─ scanpaths/ # per-trial fixation TSVs (or fixations/)
1656│ └─ *.tsv
1657└─ stimuli/
1658 ├─ word_aoi_texts/ # word boxes, one file per text
1659 │ └─ word_aoi_<text>.tsv (texts b0–b5, p0–p5)
1660 └─ aoi_texts/ # character AOIs, one file per text
1661 └─ <text>.ias
1662```
1663"""
1665_MULTIPLEYE_STRUCTURE_MD = """\
1666**Expected layout** — a MultiplEYE session set (e.g. the read-only ZH-CH-Zurich
1667sample). Identity is read from the folder + file names, so there are no id
1668columns to map:
1669```
1670<dir>/
1671├─ scanpaths/ # or fixations/
1672│ └─ <session>/ # one folder per reader session
1673│ └─ *.csv (one per stimulus page)
1674└─ stimuli_<lang>_<…>/
1675 ├─ aoi_stimuli_<…>/ # character AOIs: <stimulus>_aoi.csv
1676 ├─ config/config_*.py # font size + family (optional)
1677 ├─ stimuli_images_<lang>_* # page images (optional)
1678 └─ …_comprehension_questions_*.xlsx (optional)
1679```
1680`reading_measures/` and `participant_data.csv` (optional) enrich the load.
1681"""
1683_EYEGENBENCH_STRUCTURE_MD = """\
1684**Expected layout** — a bundle built by
1685`python scripts/prepare_eyegenbench.py --all` (or any subset of corpora). No
1686download here: build the bundle locally, then point this at where it wrote
1687the files.
1688```
1689<dir>/
1690├─ manifest.json # one entry per prepared corpus
1691└─ <corpus name>/ # e.g. PoTeC, Provo, …
1692 ├─ words.parquet
1693 ├─ fixations.parquet
1694 └─ participants.parquet
1695```
1696"""
1699def _onestop_structure_md(regime: str) -> str:
1700 """Expected-files note for one OneStop regime's dataset (DATA-63)."""
1701 from scanpath_studio import datasets
1703 listing = "\n".join(
1704 f"├─ {datasets._onestop_report_path(Path('<dir>'), kind, regime, part).name}"
1705 for part in datasets.onestop_regime_parts(regime)
1706 for kind in ("ia", "fixations")
1707 )
1708 return f"""\
1709**Expected files** — this dataset's OSF reports, two per screen, placed
1710directly in the folder (or fetched by **Download**). Only the *Paragraph*
1711reports are published per reading regime; the other screens' reports are shared
1712by the four OneStop datasets and cut to this one when read:
1713```
1714<dir>/
1715{listing}
1716```
1717"""
1720def _project_root() -> Path:
1721 """Where the *relative* data dirs (``data/OneStop`` etc.) resolve.
1723 Used to anchor the relative default data dirs and relative user-entered
1724 paths, so the "found vs. download" status resolves regardless of the
1725 process cwd (the server may run from anywhere). Computed from this module's
1726 location, not ``os.getcwd()``.
1728 ENG-59: that location is only a *project* in a source checkout. In an
1729 installed copy the folder above the package is ``site-packages`` (or the
1730 desktop bundle's ``_internal/``), so ⬇ Download wrote the corpora into the
1731 environment — orphaned by ``pip uninstall``, lost with the venv, refused on
1732 a read-only install. There they resolve under a per-user data directory
1733 instead (``SCANPATH_STUDIO_DATA_HOME`` overrides it).
1734 """
1735 checkout = Path(__file__).resolve().parent.parent
1736 if (checkout / "pyproject.toml").is_file():
1737 return checkout
1738 return _user_data_home()
1741def _user_data_home() -> Path:
1742 """Per-user home for downloaded corpora in an installed copy (ENG-59)."""
1743 override = os.environ.get("SCANPATH_STUDIO_DATA_HOME", "").strip()
1744 if override:
1745 return Path(override).expanduser()
1746 if os.name == "nt":
1747 base = os.environ.get("LOCALAPPDATA") or str(Path.home() / "AppData" / "Local")
1748 return Path(base) / "scanpath-studio"
1749 base = os.environ.get("XDG_DATA_HOME") or str(Path.home() / ".local" / "share")
1750 return Path(base) / "scanpath-studio"
1753def _default_download_folder() -> Path:
1754 """Where downloads go when nobody chose: ``SCANPATH_STUDIO_DOWNLOAD_DIR``,
1755 else ``data/`` under :func:`_project_root` (UX-184)."""
1756 configured = os.environ.get(DOWNLOAD_DIR_ENV, "").strip()
1757 if configured:
1758 return Path(_resolve_data_dir(configured))
1759 return (_project_root() / "data").resolve()
1762def download_folder() -> Path:
1763 """The folder every downloadable corpus goes into, one subfolder each (UX-184).
1765 The 🗂️ Data page's *Download folder* (a blank box means the default), else
1766 :func:`_default_download_folder`. A relative entry anchors like a Data
1767 directory does, and ``SCANPATH_DATA_ROOT`` confines it the same way."""
1768 chosen = str(st.session_state.get(DOWNLOAD_DIR_KEY) or "").strip()
1769 if chosen and local_filesystem_enabled():
1770 return Path(_resolve_data_dir(chosen))
1771 return _default_download_folder()
1774def _download_target(default_dir: str) -> str:
1775 """A downloadable corpus' default Data directory under :func:`download_folder`.
1777 The built-in defaults are ``data/<corpus>``; that ``data/`` is the download
1778 folder, so ``data/PoTeC`` becomes ``<folder>/PoTeC``. Anything else (an
1779 absolute path a test or a deployment pinned) is left as it is."""
1780 rel = Path(default_dir)
1781 if not default_dir or rel.is_absolute() or rel.parts[:1] != ("data",):
1782 return default_dir
1783 return str(download_folder().joinpath(*rel.parts[1:]))
1786# DATA-16 (security audit S2). The corpus **Data directory** box takes a
1787# free-text path from the browser, stats it, reports the result back into the
1788# page, and — via ⬇ Download — writes into it. On a local run that's just a file
1789# picker. On anything another person can reach it's a path-existence oracle plus
1790# an arbitrary-directory write, and the app has no authentication on any
1791# deployment.
1792#
1793# ENG-66: unset, it follows the server's bind address — on for a server that
1794# listens on loopback only (`scanpath-studio run`, the desktop app), off for one
1795# other machines can reach (a bare `streamlit run`, a hosted demo), the same rule
1796# as the recovery cache. `SCANPATH_LOCAL_FS=1` turns it on for a trusted lab
1797# server and `=0` forces it off. Off hides the path box, the folder picker and
1798# the download button; `SCANPATH_DATA_ROOT` then supplies the corpus location
1799# server-side. Setting `SCANPATH_DATA_ROOT` alone is also useful locally: it
1800# confines every entered path to that subtree.
1801LOCAL_FS_ENV = "SCANPATH_LOCAL_FS"
1802_LOCAL_FS_ON = frozenset({"1", "true", "yes", "on"})
1803_LOCAL_FS_OFF = frozenset({"0", "false", "no", "off"})
1804DATA_ROOT_ENV = "SCANPATH_DATA_ROOT"
1807def local_filesystem_enabled() -> bool:
1808 """Whether the user may point the app at an arbitrary local directory.
1810 ``SCANPATH_LOCAL_FS`` set to ``1``/``true``/``yes``/``on`` or
1811 ``0``/``false``/``no``/``off`` decides. Otherwise, inside a Streamlit server,
1812 it is on only when that server listens on loopback alone
1813 (:func:`persistence.server_bound_to_loopback`, ENG-66) — so a hosted
1814 deployment is safe without remembering to set anything — and outside one
1815 (the API, the CLI) it is on. Read at call time so tests can toggle it."""
1816 raw = os.environ.get(LOCAL_FS_ENV, "").strip().lower()
1817 if raw in _LOCAL_FS_ON:
1818 return True
1819 if raw in _LOCAL_FS_OFF:
1820 return False
1821 return server_bound_to_loopback() if runtime.exists() else True
1824def data_root() -> Path | None:
1825 """The configured allow-root for corpus paths, or ``None`` if unset."""
1826 raw = os.environ.get(DATA_ROOT_ENV, "").strip()
1827 return Path(raw).expanduser().resolve() if raw else None
1830def _resolve_data_dir(root: str) -> str:
1831 """Resolve a possibly-relative data dir against the project root.
1833 Absolute paths (and ``~``) are used verbatim; a relative path is joined to
1834 the project root so it resolves no matter where the server was launched from
1835 (fixes the "No data found" false-negative when cwd != repo root). A blank
1836 stays blank (the loader then shows its missing-data note).
1838 When ``SCANPATH_DATA_ROOT`` is set, the result is confined to that subtree
1839 (S2): a path resolving outside it — including via ``..`` or a symlink, since
1840 the comparison is on the *resolved* path — collapses to the root itself
1841 rather than being passed through to a stat or a download."""
1842 text = (root or "").strip()
1843 if not text:
1844 return text
1845 expanded = Path(text).expanduser()
1846 # Unchanged when no allow-root is configured: absolute paths pass through
1847 # verbatim (resolving them would rewrite a symlinked data dir in the "Found
1848 # in `…`" line), relative ones anchor to the project root.
1849 literal = expanded if expanded.is_absolute() else (_project_root() / expanded)
1850 allow_root = data_root()
1851 if allow_root is None:
1852 return str(literal if expanded.is_absolute() else literal.resolve())
1853 # The containment test is on the *resolved* path, so `..` and symlinks are
1854 # caught rather than string-matched.
1855 if not literal.resolve().is_relative_to(allow_root):
1856 return str(allow_root)
1857 return str(literal)
1860#: BUG-98 — how long a 📁 click may wait for the user to pick a folder.
1861_FOLDER_PICKER_TIMEOUT_S = 600
1863_MACOS_PICKER = 'POSIX path of (choose folder with prompt "Choose a folder")'
1864# STA + a TopMost owner form, or the dialog opens behind the browser; UTF-8 so a
1865# folder name outside the console's code page comes back intact.
1866_WINDOWS_PICKER = (
1867 "[Console]::OutputEncoding = [Text.Encoding]::UTF8; "
1868 "Add-Type -AssemblyName System.Windows.Forms; "
1869 "$d = New-Object System.Windows.Forms.FolderBrowserDialog; "
1870 "$o = New-Object System.Windows.Forms.Form -Property @{TopMost = $true}; "
1871 "if ($d.ShowDialog($o) -eq 'OK') { [Console]::Out.Write($d.SelectedPath) }"
1872)
1873_TK_PICKER = (
1874 "import tkinter as tk; from tkinter import filedialog; "
1875 "r = tk.Tk(); r.withdraw(); r.wm_attributes('-topmost', 1); "
1876 "print(filedialog.askdirectory() or '', end='')"
1877)
1880def _folder_picker_command() -> list[str] | None:
1881 """The command that shows this OS's folder dialog and prints the pick (BUG-98).
1883 The dialog runs in a child process. In-process tkinter ran on Streamlit's
1884 script thread, and macOS refuses to open a window off the main thread — it
1885 aborts the whole server (``NSWindow should only be instantiated on the main
1886 thread``), so one 📁 click took the app down. A child has its own main
1887 thread, and whatever it does cannot reach the server. The OS's own dialog
1888 comes first; tkinter is the fallback for a Linux desktop without zenity or
1889 kdialog, and never in the desktop bundle, which ships no tkinter and whose
1890 ``sys.executable`` is the app itself. ``None`` when there is none."""
1891 if sys.platform == "darwin":
1892 return ["osascript", "-e", _MACOS_PICKER] if shutil.which("osascript") else None
1893 if sys.platform == "win32":
1894 shell = shutil.which("powershell") or shutil.which("pwsh")
1895 if shell:
1896 return [shell, "-NoProfile", "-STA", "-Command", _WINDOWS_PICKER]
1897 elif shutil.which("zenity"):
1898 return ["zenity", "--file-selection", "--directory"]
1899 elif shutil.which("kdialog"):
1900 return ["kdialog", "--getexistingdirectory"]
1901 if getattr(sys, "frozen", False):
1902 return None
1903 return [sys.executable, "-c", _TK_PICKER]
1906def _pick_directory_dialog() -> str | None:
1907 """Open a native folder picker and return the chosen path, or None.
1909 Only works when the app runs on a machine with a display (a locally-run
1910 app). Returns None — and never raises — on a headless host, with no dialog
1911 to run, or on a cancelled dialog, so the text input stays the portable
1912 fallback. The dialog is a child process (:func:`_folder_picker_command`,
1913 BUG-98); this blocks until it closes, as the in-process one did.
1915 S2: refuses outright on a shared deployment. Degrading to None on a headless
1916 host was never the guarantee — on a host that *does* have a display, a remote
1917 visitor clicking 📁 pops a modal dialog on the server's own desktop and blocks
1918 the thread until someone there dismisses it."""
1919 if not local_filesystem_enabled():
1920 return None
1921 command = _folder_picker_command()
1922 if command is None:
1923 return None
1924 try:
1925 done = subprocess.run(
1926 command,
1927 capture_output=True,
1928 text=True,
1929 encoding="utf-8",
1930 errors="replace",
1931 timeout=_FOLDER_PICKER_TIMEOUT_S,
1932 check=False,
1933 )
1934 except (OSError, subprocess.SubprocessError):
1935 return None
1936 # A cancel exits non-zero (osascript, zenity, kdialog) or prints nothing.
1937 chosen = done.stdout.strip() if done.returncode == 0 else ""
1938 # osascript's POSIX path ends in "/"; keep a bare root as it is.
1939 return (chosen.rstrip("/\\") or chosen) if chosen else None
1942def _dataset_dir_input(
1943 cfg, *, default_dir: str, dir_help: str, structure_md: str, key_prefix: str
1944) -> str:
1945 """Data-location input + a native **Browse…** button + an Expected-files note.
1947 Returns the *resolved* directory (relative paths anchored to the project
1948 root, so the found/download status is correct regardless of cwd). A
1949 "📁 Browse…" button opens a native folder dialog when available (local app)
1950 and writes the pick back into the text input; it's silently skipped on a
1951 headless host, where the text box is the only control."""
1952 dir_key = f"{key_prefix}_dir"
1953 # S2: on a shared deployment the path box is a path-existence oracle, so it
1954 # isn't rendered at all — the location comes from the server's environment.
1955 if not local_filesystem_enabled():
1956 configured = str(data_root()) if data_root() else _resolve_data_dir(default_dir)
1957 cfg.caption(
1958 f"Reading from the server's configured data location: `{configured}`"
1959 )
1960 with cfg.expander("Expected files", expanded=False):
1961 st.markdown(structure_md)
1962 return configured
1963 # A prior Browse pick is applied before the widget instantiates (assigning a
1964 # widget-backed key inline after render is unreliable — see the source picker).
1965 picked = st.session_state.pop(f"{dir_key}_picked", None)
1966 if picked:
1967 st.session_state[dir_key] = picked
1968 # UX-184: a box still showing the default it was given follows a new
1969 # Download folder; one the user edited keeps what they typed.
1970 seeded_key = f"{dir_key}_default"
1971 seeded = st.session_state.get(seeded_key)
1972 if (
1973 seeded is not None
1974 and seeded != default_dir
1975 and st.session_state.get(dir_key) == seeded
1976 ):
1977 st.session_state[dir_key] = default_dir
1978 st.session_state[seeded_key] = default_dir
1979 # Seeded rather than `value=`-ed: the two writes above go through session
1980 # state, and passing both makes Streamlit warn (as for BUG-17).
1981 st.session_state.setdefault(dir_key, default_dir)
1982 text_col, browse_col = cfg.columns([4, 1])
1983 raw = text_col.text_input(
1984 "Data directory",
1985 help=dir_help,
1986 key=dir_key,
1987 # A typed path must survive a run in which this input doesn't render —
1988 # Streamlit drops an unrendered widget's key at end of run (BUG-15 /
1989 # ENG-36), and *every* one of these inputs renders only while its own
1990 # corpus is the selected source, so without this it silently forgot a
1991 # hand-typed location as soon as the user looked at another source. One
1992 # rule here rather than one call site remembering and three forgetting.
1993 persist_state="session",
1994 )
1995 # Vertical-align the button with the input (past its label).
1996 browse_col.markdown("<div style='height:1.7em'></div>", unsafe_allow_html=True)
1997 if browse_col.button(
1998 # UX-200: named for screen readers; the folder icon is all that shows.
1999 f"{ICONS['folder']} {spoken('Choose the data folder')}",
2000 wrap=True,
2001 key=f"{key_prefix}_browse",
2002 help="Browse for a folder",
2003 ):
2004 chosen = _pick_directory_dialog()
2005 if chosen:
2006 st.session_state[f"{dir_key}_picked"] = chosen
2007 st.rerun()
2008 else:
2009 cfg.caption("Folder picker unavailable here — type or paste the path.")
2010 resolved = _resolve_data_dir(raw)
2011 # UX-184: a relative entry such as the default `data/PoTeC` resolves against
2012 # the checkout, or in an installed copy against the per-user data home
2013 # (ENG-59) — on Windows `%LOCALAPPDATA%`, a folder the box never named, so a
2014 # finished download looked lost. Name the folder it actually means.
2015 if resolved and resolved != raw.strip():
2016 cfg.caption(f"Full path: `{resolved}`")
2017 with cfg.expander("Expected files", expanded=False):
2018 st.markdown(structure_md)
2019 return resolved
2022def _dataset_folder(key_prefix: str, default_dir: str) -> str:
2023 """The folder :func:`_dataset_dir_input` would resolve, without drawing it.
2025 BUG-113: the dataset table states every row's status, and a corpus that is
2026 not open has no loader running to draw its location box. This reads the
2027 same state the box keeps — a typed path, else the default, which a box still
2028 showing its old seeded default follows (UX-184) — and resolves it the same
2029 way, so the row and the loader can never disagree about where the files are.
2030 """
2031 if not local_filesystem_enabled():
2032 return str(data_root()) if data_root() else _resolve_data_dir(default_dir)
2033 dir_key = f"{key_prefix}_dir"
2034 typed = str(st.session_state.get(dir_key) or "").strip()
2035 if not typed or typed == st.session_state.get(f"{dir_key}_default"):
2036 typed = default_dir
2037 return _resolve_data_dir(typed)
2040def _potec_files_present() -> bool:
2041 from scanpath_studio import datasets
2043 root = _dataset_folder("potec", _download_target(POTEC_DEFAULT_DIR))
2044 return datasets.potec_present(root)
2047def _onestop_files_present(regime: str) -> bool:
2048 from scanpath_studio import datasets
2050 root = _dataset_folder(
2051 "onestop_public", _download_target(ONESTOP_PUBLIC_DEFAULT_DIR)
2052 )
2053 parts = datasets.onestop_regime_parts(regime)
2054 return datasets.onestop_present(root, regime=regime, parts=parts)
2057def _multipleye_files_present() -> bool:
2058 root = _dataset_folder("multipleye", MULTIPLEYE_DEFAULT_DIR)
2059 source = st.session_state.get("multipleye_fixation_source") or "scanpaths"
2060 sessions, _ = _cached_multipleye_inventory(root, source)
2061 return bool(sessions)
2064def _benchmark_files_present(dataset: str) -> bool:
2065 from scanpath_studio.eyegenbench import eyegenbench_present
2067 root = _dataset_folder("eyegenbench", EYEGENBENCH_DEFAULT_DIR)
2068 return eyegenbench_present(root, dataset)
2071# UX-7(b): session slot describing a data source the user selected but that
2072# isn't available locally. Written by `_dataset_access_status` (and the bundle
2073# sources) on the run it happens, read + cleared by `_render_dataset_unavailable`
2074# in the main area — and dropped at the start of every full run (BUG-96), so a
2075# run that returns early never passes it on. Kept out of the loader return
2076# value so the loaders can keep falling back to the demo corpus and the app
2077# stays usable.
2078_UNAVAILABLE_KEY = "_dataset_unavailable"
2079#: #374 F14 — the added dataset a `?dataset=` link opened, so it is opened once.
2080_LINK_DATASET_OPENED_KEY = "_link_dataset_opened"
2081#: UX-174: whether this run is showing the demo *in place of* the selected
2082#: corpus. Cleared at the start of every full run and set with the note above
2083#: (which is consumed before the dataset table draws), so it describes this run
2084#: on every path, early returns included; a fragment rerun of the table reads
2085#: the last full run's answer. The table reads it so the demo's rows are never
2086#: counted as that corpus' "loaded" figures.
2087_PLACEHOLDER_SHOWN_KEY = "_dataset_placeholder_shown"
2088#: DATA-48 — the corpus the bundled demo last stood in for. While it does, that
2089#: corpus' annotations are filed under the demo's name: the trials on screen are
2090#: the demo's, so a star made on them is the demo's, and must not wait under a
2091#: corpus whose own trials (once it is set up) merely share their ids.
2092_ANNOTATIONS_STANDIN_KEY = "_annotations_standin_for"
2095def annotations_owner(dataset: str) -> str:
2096 """The dataset whose annotations ``dataset``'s screen shows (DATA-48)."""
2097 if st.session_state.get(_ANNOTATIONS_STANDIN_KEY) == dataset:
2098 return DEMO_CHOICE
2099 return dataset
2102def _annotations_dataset(token: str) -> str:
2103 """The dataset the annotation store belongs to now, else ``token``."""
2104 import scanpath_studio.annotations as _annotations
2106 return _annotations.current_dataset(st.session_state) or token
2109def _file_annotations_under_shown_dataset(dataset: str) -> None:
2110 """After the load: point the annotation store at what is actually shown.
2112 ``annotations.activate_dataset`` runs before the load, when nobody knows
2113 yet whether ``dataset`` is on disk; the loader then reports the demo
2114 standing in (:data:`_PLACEHOLDER_SHOWN_KEY`). Remembering which corpus it
2115 stood in for lets the next run's first activation pick the demo straight
2116 away — one swap, not a swap and back on every run.
2117 """
2118 import scanpath_studio.annotations as _annotations
2120 if st.session_state.get(_PLACEHOLDER_SHOWN_KEY):
2121 st.session_state[_ANNOTATIONS_STANDIN_KEY] = dataset
2122 _annotations.activate_dataset(st.session_state, DEMO_CHOICE, adopt=False)
2123 elif st.session_state.get(_ANNOTATIONS_STANDIN_KEY) == dataset:
2124 st.session_state.pop(_ANNOTATIONS_STANDIN_KEY, None)
2125 _annotations.activate_dataset(st.session_state, dataset, adopt=False)
2126 # Only now is it known which dataset is shown, so only now may it adopt an
2127 # old cache's unassigned entries — which are on a built-in or public
2128 # corpus' trials, since every restored upload was already asked for its own.
2129 _annotations.adopt_unassigned(st.session_state)
2132def _note_dataset_unavailable(
2133 *,
2134 label: str,
2135 reason: str,
2136 action: str,
2137 root: str | None = None,
2138 size_hint: str = "",
2139 download: Callable[[str], None] | None = None,
2140 key_prefix: str = "",
2141) -> None:
2142 """Record that ``label`` couldn't be loaded, for the main-area empty state."""
2143 st.session_state[_PLACEHOLDER_SHOWN_KEY] = True
2144 st.session_state[_UNAVAILABLE_KEY] = dict(
2145 label=label,
2146 reason=reason,
2147 action=action,
2148 root=root,
2149 size_hint=size_hint,
2150 download=download,
2151 key_prefix=key_prefix,
2152 )
2155def _stop_download(task_key: tuple) -> None:
2156 """UX-168: Stop on a download card — the transfer ends at its next chunk."""
2157 progress.cancel(task_key)
2160def _download_with_card(
2161 slot: DeltaGenerator,
2162 download: Callable[[str], None],
2163 root: str,
2164 *,
2165 label: str,
2166 key: str,
2167) -> None:
2168 """Run ``download(root)`` under a card with a Stop button (UX-168)."""
2169 task_key = ("download", loading.session_id(), key)
2170 with loading.card(
2171 slot,
2172 key=f"download_{key}",
2173 title=f"Downloading {label}",
2174 task_key=task_key,
2175 cancel=loading.Cancel("Stop download", _stop_download, args=(task_key,)),
2176 ):
2177 download(root)
2180def _render_dataset_unavailable() -> None:
2181 """UX-7(b): a first-class "this corpus isn't here yet" state in the main area.
2183 Picking a download-on-demand corpus whose files aren't present used to change
2184 nothing visible except a line in the data-location panel — the loaders quietly
2185 fall back to the bundled demo, so the plot showed *demo* scanpaths as though
2186 the choice had taken effect. This names the dataset, says what's missing and how big the
2187 download is, offers the action inline, and states plainly that the demo is
2188 what's on screen meanwhile.
2189 """
2190 note = st.session_state.pop(_UNAVAILABLE_KEY, None)
2191 if not note:
2192 return
2193 download = note["download"]
2194 size = f" · {note['size_hint']}" if note["size_hint"] else ""
2195 # One panel, not four stacked blocks. The first version was an st.warning
2196 # banner + an st.caption + a body paragraph + a button — three type colours
2197 # and three background colours for what is a single message.
2198 with st.container(border=True, key="dataset_unavailable_panel"):
2199 st.markdown(
2200 f"#### {ICONS['missing_bundle']} {note['label']} isn't here yet\n"
2201 f"{note['reason'][:1].upper()}{note['reason'][1:].rstrip('.')} — "
2202 "**showing the bundled demo** meanwhile."
2203 )
2204 details = [f"{note['action'].rstrip('.')}{size}"]
2205 if note["root"]:
2206 # UX-184: say where a download will land, not only where it looked.
2207 verb = "Downloads to" if download is not None else "Looking in"
2208 details.append(f"{verb} `{note['root']}`")
2209 st.markdown("\n".join(f"- {line}" for line in details))
2210 if download is None:
2211 return
2212 clicked = st.button(
2213 "⬇ Download now",
2214 key=f"{note['key_prefix']}_download_main",
2215 type="primary",
2216 )
2217 download_slot = st.empty()
2218 if clicked:
2219 # UX-166: the download is a wait of its own. Take the dataset card
2220 # and the skeleton down first, or they hide this panel and its
2221 # spinner while describing a step that isn't what is running.
2222 loading.release_page()
2223 try:
2224 _download_with_card(
2225 download_slot,
2226 download,
2227 note["root"],
2228 label=note["label"],
2229 key=f"{note['key_prefix']}_main",
2230 )
2231 except (OSError, ValueError) as exc:
2232 st.error(
2233 f"Download failed: {exc}\n\nOffline? Download the files on "
2234 "another computer and copy them into the folder named above."
2235 )
2236 return
2237 st.rerun()
2240def _dataset_access_status(
2241 cfg,
2242 *,
2243 root: str,
2244 present: bool,
2245 download: Callable[[str], None] | None = None,
2246 size_hint: str = "",
2247 key_prefix: str = "",
2248 label: str = "This dataset",
2249) -> bool:
2250 """Found / missing status + an optional **Download** button.
2252 Returns ``True`` when the corpus is present on disk (ready to load). When
2253 it's missing and ``download`` is given, renders a Download button that
2254 fetches the files then reruns — the next run loads from disk with no
2255 re-download (replaces the old always-on "Download if missing" checkbox, so
2256 an already-downloaded corpus never re-checks the network).
2258 A missing corpus is also recorded for the main-area empty state (UX-7): the
2259 data-location line alone is easy to miss when the plot keeps rendering demo
2260 data.
2261 """
2262 if present:
2263 cfg.success(f"Found in `{root}`")
2264 return True
2265 if download is None:
2266 cfg.warning(
2267 f"No data found in `{root}` — point at a folder with the files above."
2268 )
2269 _note_dataset_unavailable(
2270 label=label,
2271 reason="its files aren't in the folder you pointed at.",
2272 action=f"Set **Data location** on the {ICONS['view_data']} Data Management page to a folder "
2273 "holding the files listed under **Expected files**.",
2274 root=root,
2275 )
2276 return False
2277 cfg.info(
2278 f"Not downloaded yet{f' ({size_hint})' if size_hint else ''}. "
2279 f"**Download** saves it to `{root}`."
2280 )
2281 # S2: fetching writes tens-to-hundreds of MB into a browser-supplied path. On
2282 # a shared deployment that's a remote visitor filling the server's disk, so
2283 # the corpus has to be placed by whoever runs it.
2284 if not local_filesystem_enabled():
2285 cfg.caption(
2286 "This server doesn't download datasets. Open it in the desktop app "
2287 "or a pip install, where it downloads in one click. Running this "
2288 "server yourself? Place the corpus in its data location, or, on a "
2289 "trusted network, start it with `SCANPATH_LOCAL_FS=1`."
2290 )
2291 _note_dataset_unavailable(
2292 label=label,
2293 reason="this server doesn't download datasets.",
2294 action=f"Open it in the [desktop app]({CITATION['desktop_url']}) ↗ "
2295 "or a pip install (`pip install scanpath-studio`), where it "
2296 "downloads in one click",
2297 )
2298 return False
2299 _note_dataset_unavailable(
2300 label=label,
2301 reason="it hasn't been downloaded yet.",
2302 action="Download it once; later loads read it from disk",
2303 root=root,
2304 size_hint=size_hint,
2305 download=download,
2306 key_prefix=key_prefix,
2307 )
2308 clicked = cfg.button("⬇ Download", key=f"{key_prefix}_download", type="primary")
2309 download_slot = cfg.empty()
2310 if clicked:
2311 # UX-166: as for the main area's ⬇ Download now — the dataset card must
2312 # not cover the download with a step that isn't what is running.
2313 loading.release_page()
2314 try:
2315 _download_with_card(
2316 download_slot, download, root, label=label, key=key_prefix
2317 )
2318 except (OSError, ValueError) as exc:
2319 cfg.error(f"Download failed: {exc}. Check your connection and retry.")
2320 return False
2321 st.rerun()
2322 return False
2325# UX-166: the dataset card lists this step.
2326@st.cache_data(show_spinner=False)
2327def _cached_potec_raw_frames(root: str) -> tuple[pd.DataFrame, pd.DataFrame]:
2328 """Cached raw PoTeC frames (pre-normalization) — the full corpus.
2330 Returns the same shape as an upload: raw frames the normal
2331 auto-detect → normalize → harmonize pipeline then handles. Cached on the
2332 directory so re-runs (toggling viz controls) don't re-read the files. Loads
2333 every reader × text (75 × 12); narrow the trial pool with **Narrow by**."""
2334 from scanpath_studio.datasets import potec_raw_frames
2336 return stamp_source(potec_raw_frames(root))
2339def _load_potec_source(
2340 options_host=None, location_host=None
2341) -> tuple[pd.DataFrame, pd.DataFrame]:
2342 """Sidebar controls + loader for the PoTeC corpus data source.
2344 PoTeC can't be loaded through the generic Upload flow (trial/word ids live
2345 in filenames, fixation coordinates come from a separate character-AoI
2346 file), so this dedicated source wraps ``datasets.potec_raw_frames``. The
2347 returned raw frames go through the same normalization as an upload, so the
2348 Column-mapping panels still appear and stay overridable. The whole
2349 corpus loads — narrow it with the **Narrow by** trial filters.
2351 ``options_host`` / ``location_host`` are the DATA-9 sub-slots; PoTeC
2352 has no source options, so only the data-location slot is used (defaults to a
2353 standalone expander when called without slots).
2354 """
2355 from scanpath_studio import datasets
2357 loc = location_host if location_host is not None else st.container()
2358 root = _dataset_dir_input(
2359 loc,
2360 default_dir=_download_target(POTEC_DEFAULT_DIR),
2361 dir_help="The folder that holds the PoTeC files, or that **Download** "
2362 "fills. A clone of github.com/DiLi-Lab/PoTeC works.",
2363 structure_md=_POTEC_STRUCTURE_MD,
2364 key_prefix="potec",
2365 )
2366 ready = _dataset_access_status(
2367 loc,
2368 root=root,
2369 present=datasets.potec_present(root),
2370 download=datasets.download_potec,
2371 size_hint="~45 MB",
2372 key_prefix="potec",
2373 label="PoTeC — Potsdam Textbook Corpus",
2374 )
2375 if not ready:
2376 return load_sample_data()
2377 try:
2378 return _cached_potec_raw_frames(root)
2379 except (FileNotFoundError, ValueError, OSError) as exc:
2380 loc.error(
2381 f"Couldn't load PoTeC from `{root}`: {exc}. Check it holds the "
2382 "**Expected files**."
2383 )
2384 return pd.DataFrame(), pd.DataFrame()
2387# UX-166: the dataset card lists this step.
2388@st.cache_data(show_spinner=False)
2389def _cached_multipleye_raw_frames(
2390 root: str, fixation_source: str
2391) -> tuple[pd.DataFrame, pd.DataFrame]:
2392 """Cached raw MultiplEYE frames (pre-normalization) — the full session set.
2394 Same shape as an upload — the normal auto-detect → normalize → harmonize
2395 pipeline then handles them — cached on the selection so re-runs (toggling
2396 viz controls) don't re-read the files. Loads every session × stimulus;
2397 narrow the trial pool with **Narrow by**."""
2398 from scanpath_studio.datasets import multipleye_raw_frames
2400 return stamp_source(multipleye_raw_frames(root, fixation_source=fixation_source))
2403@st.cache_data(show_spinner=False)
2404def _cached_multipleye_inventory(
2405 root: str, fixation_source: str
2406) -> tuple[tuple[str, ...], tuple[str, ...]]:
2407 from scanpath_studio.datasets import multipleye_inventory
2409 return multipleye_inventory(root, fixation_source=fixation_source)
2412def _load_multipleye_source(
2413 options_host=None, location_host=None
2414) -> tuple[pd.DataFrame, pd.DataFrame]:
2415 """Sidebar controls + loader for the MultiplEYE corpus data source.
2417 MultiplEYE can't be loaded through the generic Upload flow (participant /
2418 trial / stimulus live only in the folder + file names), so this dedicated
2419 source wraps ``datasets.multipleye_raw_frames``. The returned raw frames go
2420 through the same normalization as an upload, so the Column-mapping
2421 panels still appear and stay overridable. The whole session set loads —
2422 narrow it with the **Narrow by** trial filters.
2424 ``options_host`` / ``location_host`` are the DATA-9 sub-slots (the
2425 fixation-source radio above, the data location below); default to their own
2426 expanders when called standalone.
2427 """
2428 opt = options_host if options_host is not None else st.container()
2429 loc = location_host if location_host is not None else st.container()
2430 fixation_source = opt.radio(
2431 "Fixation source",
2432 options=["scanpaths", "fixations"],
2433 key="multipleye_fixation_source",
2434 help="scanpaths/ fixations are pre-tagged with page + word index "
2435 "(richer); fixations/ are raw onset/duration/x/y with no word linkage.",
2436 )
2437 root = _dataset_dir_input(
2438 loc,
2439 default_dir=MULTIPLEYE_DEFAULT_DIR,
2440 dir_help="Folder holding a MultiplEYE session set, e.g. the read-only "
2441 "ZH-CH-Zurich sample.",
2442 structure_md=_MULTIPLEYE_STRUCTURE_MD,
2443 key_prefix="multipleye",
2444 )
2445 try:
2446 sessions_all, _ = _cached_multipleye_inventory(root, fixation_source)
2447 except (FileNotFoundError, OSError):
2448 sessions_all = ()
2449 # MultiplEYE ships no public download URL — present means the local folder
2450 # holds a recognizable session set, otherwise fall back to the demo.
2451 ready = _dataset_access_status(
2452 loc,
2453 root=root,
2454 present=bool(sessions_all),
2455 key_prefix="multipleye",
2456 label="MultiplEYE — multilingual reading",
2457 )
2458 if not ready:
2459 return load_sample_data()
2460 try:
2461 return _cached_multipleye_raw_frames(root, fixation_source)
2462 except (FileNotFoundError, ValueError, OSError) as exc:
2463 loc.error(f"Couldn't load MultiplEYE from `{root}`: {exc}")
2464 return pd.DataFrame(), pd.DataFrame()
2467# UX-166: the dataset card lists this step.
2468@st.cache_data(show_spinner=False)
2469def _cached_onestop_raw_frames(
2470 root: str, regime: str, parts: tuple[str, ...], variant: str
2471) -> tuple[pd.DataFrame, pd.DataFrame]:
2472 """Cached raw OneStop frames (pre-normalization) for a regime + parts + variant.
2474 Cached on (root, regime, parts, variant) so toggling viz controls doesn't
2475 re-read the reports. The reports are present by the time this runs (the
2476 loader's Download button fetched them, or the lacclab export is local), so
2477 it never touches the network."""
2478 from scanpath_studio.datasets import onestop_raw_frames
2480 return stamp_source(
2481 onestop_raw_frames(root, regime=regime, parts=list(parts), variant=variant)
2482 )
2485def _load_onestop_regime_source(
2486 options_host=None, location_host=None, *, regime: str
2487) -> tuple[pd.DataFrame, pd.DataFrame]:
2488 """Loader for one OneStop regime's dataset — every part, from OSF (DATA-63).
2490 Each reading regime is its own entry in the dataset table, so there are no
2491 source options to pick: the regime is the dataset, and it holds all of that
2492 regime's parts (`datasets.onestop_regime_parts`), each part its own trial.
2493 The reports are the public OSF release, downloaded once into one folder the
2494 four regimes share (the parts other than Paragraph are the same files for
2495 all of them). They share the bundled demo's schema, so the raw frames go
2496 through the normal normalization pipeline and the Column-mapping panels
2497 still appear. Distinct from the env-var "OneStop server bundle" source.
2499 ``options_host`` is unused (kept for the registry's loader signature);
2500 ``location_host`` is the DATA-9 data-location sub-slot.
2501 """
2502 from scanpath_studio import datasets
2504 loc = location_host if location_host is not None else st.container()
2505 parts = datasets.onestop_regime_parts(regime)
2506 root = _dataset_dir_input(
2507 loc,
2508 # UX-184: under the one Download folder every public corpus shares.
2509 default_dir=_download_target(ONESTOP_PUBLIC_DEFAULT_DIR),
2510 dir_help="Folder to download the OneStop reports into (kept there, so "
2511 "only the first load fetches them). The four OneStop datasets can share it.",
2512 structure_md=_onestop_structure_md(regime),
2513 key_prefix="onestop_public",
2514 )
2515 present = datasets.onestop_present(root, regime=regime, parts=parts)
2516 ready = _dataset_access_status(
2517 loc,
2518 root=root,
2519 present=present,
2520 download=lambda r: datasets.download_onestop(r, regime=regime, parts=parts),
2521 size_hint=f"{2 * len(parts)} OSF reports, hundreds of MB each",
2522 # Per regime: two regimes' Download buttons must not share a key.
2523 key_prefix=f"onestop_{regime}",
2524 label=picker_name_for(ONESTOP_REGIME_CHOICES[regime]),
2525 )
2526 if not ready:
2527 return load_sample_data()
2528 try:
2529 return _cached_onestop_raw_frames(root, regime, tuple(parts), "public")
2530 except (FileNotFoundError, ValueError, OSError) as exc:
2531 loc.error(
2532 f"Couldn't load OneStop from `{root}`: {exc}. Check it holds the "
2533 "**Expected files**."
2534 )
2535 return pd.DataFrame(), pd.DataFrame()
2538# UX-166: the dataset card lists this step.
2539@st.cache_data(show_spinner=False)
2540def _cached_eyegenbench_raw_frames(
2541 root: str, dataset: str
2542) -> tuple[pd.DataFrame, pd.DataFrame]:
2543 """Cached raw ``(words, fixations)`` for one EyeGenBench corpus.
2545 Cached on ``(root, dataset)`` so re-runs (toggling viz controls) don't
2546 re-read the Parquet files. Keyed on plain strings, not the manifest entry
2547 dict, so the cache survives an unrelated manifest re-read."""
2548 from scanpath_studio.eyegenbench import eyegenbench_raw_frames
2550 return stamp_source(eyegenbench_raw_frames(root, dataset=dataset))
2553# A malformed manifest (an entry with no `name`, a `datasets` value that isn't a
2554# list of objects) must degrade to a load error, not crash the app: `KeyError`
2555# is in here because it escapes the usual IO triple and every reader of a
2556# manifest reads entry keys (M7).
2557_MANIFEST_ERRORS = (FileNotFoundError, ValueError, OSError, KeyError)
2560def added_benchmark_datasets() -> tuple:
2561 """Manifest entries for the benchmark corpora the user added — none yet.
2563 DATA-55 retired automatic discovery. The app used to list every corpus in a
2564 bundle it found on disk (``data/EyeGenBench``, or a folder typed into a
2565 "set up" entry), which put data in the picker that nobody had chosen, so a
2566 corpus is now listed only because someone added it. The flow that adds one —
2567 choose a folder, scan it, pick the corpora — is DATA-56. Until it lands this
2568 is empty, and the per-corpus machinery it feeds (the registry entry, the
2569 loader, the geometry badge, the share-link slug, Compare and the code
2570 snippet) is reached only by tests, which replace this function.
2571 """
2572 return ()
2575# geometry_source values are eyegenbench_geometry.py's GEOMETRY_REAL /
2576# _RECONSTRUCTED / _SYNTHESIZED (that module owns the tiering; not touched
2577# here). Surfaced on each corpus' entry so a user can tell which they're looking
2578# at rather than trusting a blanket claim in the description.
2579_EYEGENBENCH_GEOMETRY_BADGES = {
2580 "real": f"{ICONS['geometry_real']} **Real** screen geometry — measured word boxes.",
2581 "reconstructed": f"{ICONS['geometry_reconstructed']} **Reconstructed** geometry — no measured boxes for "
2582 "this corpus; derived from its documented display setup.",
2583 "synthesized": f"{ICONS['geometry_synthesized']} **Synthesized** geometry — no measured boxes or "
2584 "documented display setup; a default layout was assumed.",
2585}
2588def _geometry_coverage_note(entry) -> str:
2589 """How much of a ``real`` corpus is actually measured, or ``""`` (R34).
2591 The single source for that qualifier: the badge and the picker description
2592 print it in the same panel, one line apart, so two spellings of the rule is
2593 how one of them ends up claiming uniform geometry the other has just denied
2594 (M8). Empty when the corpus is uniform, or when the tier is one that already
2595 says *no* measured boxes.
2597 ``n_texts`` is missing from no real manifest, but when it is the note goes
2598 vague rather than silent (M11): "some texts aren't measured" is worse copy
2599 than a count and a better claim than a confident, possibly-wrong "Real".
2601 The counts are read through `eyegenbench.entry_count`, which is also what
2602 keeps a hand-mangled manifest from raising out of the *picker build* — this
2603 runs for every added corpus via `_benchmark_description` (N1). An
2604 unreadable count lands in the same vaguer wording as an absent one: it is
2605 the R34-honest answer either way, and it is never worth taking the source
2606 list down over a typo in a number.
2607 """
2608 from scanpath_studio.eyegenbench import entry_count
2610 if str(entry.get("geometry_source") or "").strip() != "real":
2611 return ""
2612 missing = entry_count(entry, "paragraphs_without_real_boxes")
2613 if missing is not None and missing <= 0:
2614 return ""
2615 total = entry_count(entry, "n_texts")
2616 covered = (
2617 f"measured word boxes for {max(total - missing, 0)} of {total} texts"
2618 if missing is not None and total
2619 else "measured word boxes for some but not all texts"
2620 )
2621 return f"{covered}; the rest fall back to reconstructed layout"
2624def geometry_badge(entry) -> str:
2625 """The one-line geometry-provenance badge for a manifest entry (R34).
2627 `eyegenbench_geometry.py` promotes a whole corpus to ``real`` when **any**
2628 paragraph has measured word boxes — the scalar means *best tier achieved*,
2629 and it keeps that meaning (the per-word column already refines it, and
2630 changing it would ripple into the manifest contract, the CLI and the API).
2631 What must not happen is a *rendering* that implies uniformity: a corpus with
2632 one measured text in a thousand would otherwise read as "✅ Real — measured
2633 word boxes". So whenever ``paragraphs_without_real_boxes`` is non-zero the
2634 real badge says how many texts it actually covers, and plain "Real" is
2635 reserved for full coverage.
2637 The reconstructed / synthesized badges need no such qualifier: they already
2638 say *no* measured boxes, which is exactly what a non-zero count means there.
2639 """
2640 source = str(entry.get("geometry_source") or "").strip()
2641 if not source:
2642 return ""
2643 badge = _EYEGENBENCH_GEOMETRY_BADGES.get(source)
2644 if badge is None:
2645 return f"Screen geometry: {source}"
2646 if note := _geometry_coverage_note(entry):
2647 badge = f"{ICONS['geometry_real']} **Real** screen geometry — {note}."
2648 try:
2649 recorded_y = float(entry.get("recorded_fixation_y_fraction", 0.0))
2650 except (TypeError, ValueError):
2651 recorded_y = 0.0
2652 if recorded_y >= 0.9995:
2653 y_note = "Recorded fixation y."
2654 elif recorded_y > 0:
2655 y_note = (
2656 f"Recorded fixation y for {recorded_y:.0%}; other y positions use "
2657 "word-box centers."
2658 )
2659 else:
2660 y_note = "Fixation y uses word-box centers."
2661 return f"{badge} {y_note}"
2664def benchmark_corpus_label(name: str) -> str:
2665 """The registry key for a prepared corpus named ``name``."""
2666 return f"{name}{BENCHMARK_LABEL_SUFFIX}"
2669def picker_name_for(choice: str, registry: dict | None = None) -> str:
2670 """Exactly the name the **Data source** picker renders for ``choice``.
2672 Anything that tells a user to "select X" must quote this, not the registry
2673 key. The two differ: the picker shows the entry's `short`, with a (WIP)
2674 marker on top of it for a benchmark corpus, so *"Provo — harmonised benchmark
2675 corpus"* is offered as *"Provo (WIP)"*. A remedy naming a string that appears
2676 nowhere in the list is worse than no remedy — the reader hunts for it and
2677 concludes the app is broken.
2679 Pass ``registry`` when formatting a list of options: the added corpora can
2680 change at runtime, so one run must format every option against
2681 **one** snapshot (M6). Re-resolving per option lets an option's rendered text
2682 change underneath a widget mid-run, and Streamlit finds the selected value's
2683 formatted form no longer among its own options.
2684 """
2685 spec = (registry if registry is not None else public_dataset_registry()).get(choice)
2686 if spec is None:
2687 return choice
2688 name = str(spec.get("short") or choice)
2689 return f"{name}{BENCHMARK_WIP_SUFFIX}" if spec_is_benchmark(spec) else name
2692def mark_wip_if_benchmark(choice: str) -> str:
2693 """``choice`` with the (WIP) marker when it names a harmonised corpus.
2695 The marker has to reach **every** picker that offers these corpora, not just
2696 the data-source one: Compare's *Scanpath B from* selectbox can load a corpus
2697 as scanpath B, and a user who only ever meets it there would publish a
2698 comparison against an unfinished feature without being told. Display-only in
2699 both places, and the same predicate decides both.
2700 """
2701 spec = public_dataset_registry().get(choice)
2702 return f"{choice}{BENCHMARK_WIP_SUFFIX}" if spec_is_benchmark(spec) else choice
2705def spec_is_benchmark(spec) -> bool:
2706 """True for a registry entry this feature owns: a prepared benchmark corpus.
2708 Dispatches on the entry's own `benchmark_dataset` field (set by
2709 `_benchmark_registry_entries`) — the same discriminator `compare_source`
2710 uses, and deliberately **not** on the label's text: PoTeC and OneStop each
2711 ship natively *and* harmonised, so a substring test on the label would sweep
2712 the native entries in too.
2713 """
2714 if not isinstance(spec, dict):
2715 return False
2716 return bool(spec.get("benchmark_dataset"))
2719def _benchmark_short_name(name: str) -> str:
2720 """The picker's display name for a prepared corpus.
2722 PoTeC and OneStop ship **both** natively in this app and in the benchmark
2723 set, and the user wants both kept: the harmonised versions are what make
2724 cross-corpus comparison possible, which is the point of a harmonised suite.
2725 They are distinguished by the property that actually differs — one is the
2726 publisher's own release, the other a re-derived harmonisation — rather than
2727 by naming the pipeline (which is being extracted into its own repository, so
2728 anything user-visible carrying its name would need renaming later).
2730 The overlap is computed against the static built-ins rather than hard-coded,
2731 so adding a native corpus that a bundle also carries disambiguates itself.
2732 """
2733 natives = {
2734 str(spec.get("short") or "").lower()
2735 for spec in PUBLIC_DATASET_REGISTRY.values()
2736 }
2737 if name.lower() in natives:
2738 return f"{name}{BENCHMARK_SHORT_SUFFIX}"
2739 return name
2742def _benchmark_size_caption(entry) -> str:
2743 """``"84 participants · 55 texts · 219,556 fixations"`` from a manifest entry.
2745 Counts that don't parse are simply left out of the caption — via the same
2746 `entry_count` the geometry note reads, rather than a second hand-rolled
2747 ``try`` (N1).
2748 """
2749 from scanpath_studio.eyegenbench import entry_count
2751 parts = []
2752 for key, singular in (
2753 ("n_readers", "participant"),
2754 ("n_texts", "text"),
2755 ("n_fixations", "fixation"),
2756 ):
2757 if count := entry_count(entry, key):
2758 parts.append(f"{count:,} {singular}{'' if count == 1 else 's'}")
2759 return " · ".join(parts)
2762def _benchmark_description(entry, *, harmonised_overlap: bool) -> str:
2763 """The one-line description under a prepared corpus' picker entry.
2765 Carries the provenance ("EyeGenBench" belongs here, not in the label) and —
2766 for a corpus this app also ships natively — the fidelity difference, which is
2767 the whole reason both are offered.
2768 """
2769 from scanpath_studio.eyegenbench import entry_name
2771 name = entry_name(entry)
2772 lead = (
2773 f"{name}, re-derived by EyeGenBench into the benchmark's common "
2774 f"schema — the same corpus as this app's own {name} entry, prepared for "
2775 "cross-corpus comparison rather than the publisher's own geometry."
2776 if harmonised_overlap
2777 else f"{name} — a public reading corpus, harmonized by EyeGenBench to "
2778 "one common schema."
2779 )
2780 tail = []
2781 if source := str(entry.get("geometry_source") or "").strip():
2782 # Same qualifier as the badge rendered beside this (M8) — a bare
2783 # "Screen geometry: real" next to "measured word boxes for 9 of 12
2784 # texts" is the overclaim R34 exists to prevent, one line away from
2785 # the fix.
2786 note = _geometry_coverage_note(entry)
2787 tail.append(f"Screen geometry: {source}" + (f" — {note}" if note else ""))
2788 if license_ := str(entry.get("license") or "").strip():
2789 tail.append(f"License: {license_}")
2790 if citation := str(entry.get("citation") or "").strip():
2791 tail.append(citation)
2792 return lead + (" " + ". ".join(tail) + "." if tail else "")
2795def _benchmark_dir_input(loc) -> str:
2796 """The shared bundle-directory input, rendered by every benchmark entry.
2798 One session key (`eyegenbench_dir`) across all of them: the corpora live in
2799 one prepared bundle, so pointing any entry somewhere else moves them all.
2800 """
2801 return _dataset_dir_input(
2802 loc,
2803 default_dir=EYEGENBENCH_DEFAULT_DIR,
2804 dir_help="Folder holding a prepared benchmark bundle. Build one with "
2805 "`python scripts/prepare_eyegenbench.py --all` — there is no download "
2806 "from here.",
2807 structure_md=_EYEGENBENCH_STRUCTURE_MD,
2808 key_prefix="eyegenbench",
2809 )
2812def _load_benchmark_source(
2813 options_host=None, location_host=None, *, dataset: str = ""
2814) -> tuple[pd.DataFrame, pd.DataFrame]:
2815 """Location controls + loader for **one** prepared benchmark corpus.
2817 A peer of `_load_potec_source` / `_load_multipleye_source`: one entry, one
2818 corpus, no sub-picker. The returned raw frames go through the same
2819 normalization as an upload, so the Column-mapping panels still appear.
2820 """
2821 from scanpath_studio.eyegenbench import entry_name, eyegenbench_present
2823 opt = options_host if options_host is not None else st.container()
2824 loc = location_host if location_host is not None else st.container()
2825 root = _benchmark_dir_input(loc)
2826 try:
2827 present = eyegenbench_present(root, dataset)
2828 except _MANIFEST_ERRORS:
2829 present = False
2830 ready = _dataset_access_status(
2831 loc,
2832 root=root,
2833 present=present,
2834 key_prefix="eyegenbench",
2835 # The name the picker shows for this corpus, not a hand-built one. The
2836 # "(harmonised benchmark)" suffix is added only when a native entry of
2837 # the same name exists (`_benchmark_short_name`), so hardcoding it here
2838 # made the empty-state call Provo "Provo (harmonised benchmark)" while
2839 # the picker called it "Provo (WIP)" — two names, neither matching.
2840 label=picker_name_for(benchmark_corpus_label(dataset)),
2841 )
2842 if not ready:
2843 return load_sample_data()
2844 entry = next(
2845 (e for e in added_benchmark_datasets() if entry_name(e) == dataset),
2846 None,
2847 )
2848 if entry and (badge := geometry_badge(entry)):
2849 opt.caption(badge)
2850 try:
2851 return _cached_eyegenbench_raw_frames(root, dataset)
2852 except _MANIFEST_ERRORS as exc:
2853 loc.error(f"Couldn't load '{dataset}' from `{root}`: {exc}")
2854 return pd.DataFrame(), pd.DataFrame()
2857# Registry behind the "Public datasets" source: label → loader (renders its own
2858# source options and returns raw, pre-normalization frames), the corpus'
2859# presentation-monitor size (canvas default for true-to-scale rendering; None to
2860# estimate from data extents), and a little presentation metadata (a short name
2861# for the picker, plus language / size / description / home link shown as a
2862# caption). To add a corpus: write a loader in datasets.py, wrap it in a
2863# `_load_*_source` function above, and add one entry here — the
2864# searchable picker scales as the catalogue grows.
2865#: The MultiplEYE entry's registry label, named because DATA-54's beta gate
2866#: (`constants.multipleye_enabled`) has to find it.
2867MULTIPLEYE_PUBLIC_CHOICE = "MultiplEYE — multilingual reading (ZH-CH sample)"
2869#: DATA-63 — a regime dataset's ``?source=`` token → its registry label.
2870ONESTOP_REGIME_TOKEN_CHOICES = {
2871 token: ONESTOP_REGIME_CHOICES[regime]
2872 for regime, token in ONESTOP_REGIME_SOURCE_TOKENS.items()
2873}
2875#: DATA-63 — what each OneStop regime's dataset says about itself.
2876_ONESTOP_REGIME_DESCRIPTIONS = {
2877 "ordinary": "native English speakers reading Guardian articles for "
2878 "comprehension, without seeing the question first.",
2879 "information_seeking": "native English speakers reading Guardian articles "
2880 "after seeing the question they will answer.",
2881 "repeated": "native English speakers reading a paragraph for the second "
2882 "time, without seeing the question first.",
2883 "information_seeking_repeated": "native English speakers reading a "
2884 "paragraph for the second time, after seeing the question.",
2885}
2888#: DATA-65 — each regime's figures, counted from the reports at the OSF version
2889#: `datasets` pins, through this app's own load (every part, as the regime's
2890#: dataset loads them) and `_dataset_counts`, on 2026-10-02. The corpus
2891#: publishes figures for the whole release only, so these are measured, not
2892#: quoted; the pin is what makes them stay true. A regime not listed has not been
2893#: counted yet and fills in once it is opened. Recount after a pipeline change
2894#: that moves trials, texts or screens (DATA-63 did: it made each part a screen).
2895_ONESTOP_REGIME_COUNTS: dict[str, dict[str, int]] = {
2896 "ordinary": {
2897 "Participants": 180,
2898 "Texts": 330,
2899 "Trials": 10078,
2900 "Screens": 52369,
2901 "Words": 2096092,
2902 "Fixations": 2085412,
2903 },
2904 "information_seeking": {
2905 "Participants": 180,
2906 "Texts": 330,
2907 "Trials": 10080,
2908 "Screens": 62459,
2909 "Words": 2191989,
2910 "Fixations": 1944158,
2911 },
2912 "repeated": {
2913 "Participants": 180,
2914 "Texts": 324,
2915 "Trials": 1944,
2916 "Screens": 10080,
2917 "Words": 399394,
2918 "Fixations": 296202,
2919 },
2920 "information_seeking_repeated": {
2921 "Participants": 180,
2922 "Texts": 324,
2923 "Trials": 1944,
2924 "Screens": 12023,
2925 "Words": 416956,
2926 "Fixations": 260190,
2927 },
2928}
2931def _onestop_regime_entry(regime: str) -> dict:
2932 """The registry entry for one OneStop regime's dataset (DATA-63).
2934 ``published_counts`` are this regime's own, measured (DATA-65,
2935 :data:`_ONESTOP_REGIME_COUNTS`) — never the whole release's, which no single
2936 regime holds.
2937 """
2938 label = ONESTOP_REGIME_LABELS[regime]
2939 counts = _ONESTOP_REGIME_COUNTS.get(regime)
2940 extra = (
2941 dict(
2942 published_counts=counts,
2943 published_counts_source=(
2944 "Counted from the OSF reports at the version this release pins, "
2945 "by this app's own load of every part of the regime."
2946 ),
2947 )
2948 if counts
2949 else {}
2950 )
2951 return dict(
2952 **extra,
2953 loader=partial(_load_onestop_regime_source, regime=regime),
2954 # BUG-113: the dataset table's Status, for a row that is not open.
2955 files_present=partial(_onestop_files_present, regime),
2956 downloadable=True,
2957 # OneStop presentation monitor (full-screen px coords). Sourced in
2958 # `eyegenbench_geometry.DISPLAY_SPECS["onestop"]` — Berzak et al. 2025,
2959 # Sci Data 12:1995, Methods → Apparatus, which states the Dell U2715H
2960 # at 2560 px × 1440 px over a 597 mm × 336 mm display area.
2961 monitor=(2560, 1440),
2962 # Distinct per regime: `short` is the stable identifier a Share link's
2963 # corpus slug and the code snippet are derived from.
2964 short=f"OneStop · {label}",
2965 onestop_regime=regime,
2966 language="English (L1)",
2967 size=f"{label} · all screens, from the OSF release",
2968 description=f"OneStop Eye Movements, {label.lower()} — "
2969 f"{_ONESTOP_REGIME_DESCRIPTIONS[regime]}",
2970 link="https://github.com/lacclab/OneStop-Eye-Movements",
2971 )
2974PUBLIC_DATASET_REGISTRY: dict = {
2975 "PoTeC — Potsdam Textbook Corpus": dict(
2976 loader=_load_potec_source,
2977 files_present=_potec_files_present, # BUG-113
2978 downloadable=True,
2979 # The schema `load_potec` uses, so the app's Trial ID is the headless
2980 # one: reader + text, not the text name every reader shares.
2981 declared_schemas=(POTEC_WORD_SCHEMA, POTEC_FIX_SCHEMA),
2982 monitor=(1680, 1050), # DELL P2210
2983 short="PoTeC",
2984 language="German",
2985 size="75 participants · 12 texts",
2986 description="Potsdam Textbook Corpus — German readers, experts and "
2987 "novices, reading biology and physics textbook passages.",
2988 link="https://github.com/DiLi-Lab/PoTeC",
2989 # DATA-36: what the row shows before anyone opens it.
2990 published_counts={
2991 "Participants": 75,
2992 "Texts": 12,
2993 "Trials": 900,
2994 "Words": 142125,
2995 "Fixations": 404420,
2996 },
2997 published_counts_source=(
2998 "PoTeC's own README for participants, texts and trials. Words and "
2999 "fixations were measured from the released corpus; the fixation "
3000 "total agrees with the harmonized bundle's manifest to the row."
3001 ),
3002 # Word boxes come from the corpus' own `.ias` character files, but the
3003 # release discards the recorded screen (x, y) — `datasets._potec_fixations`
3004 # places each fixation at the centre of the character it names. UX-177:
3005 # the one provenance fact that changes how a figure is read, so it is
3006 # the one said on the Data page.
3007 reading_note="Fixation positions are reconstructed, not recorded: "
3008 "PoTeC's release keeps no screen coordinates, so each fixation is drawn "
3009 "at the center of the character it landed on.",
3010 ),
3011 MULTIPLEYE_PUBLIC_CHOICE: dict(
3012 loader=_load_multipleye_source,
3013 # BUG-113. No download: MultiplEYE is read from a local session set.
3014 files_present=_multipleye_files_present,
3015 monitor=(1920, 1080), # MultiplEYE physical screen (coords offset to it)
3016 short="MultiplEYE",
3017 language="Multilingual (ZH-CH sample)",
3018 size="local session set",
3019 description="MultiplEYE multilingual eye-tracking-while-reading — the "
3020 "read-only Zurich Chinese sample, loaded from a local folder.",
3021 link="https://multipleye.eu/",
3022 # DATA-36: no published figures on purpose. This source reads whichever
3023 # session folders are on the machine it runs on, so there is no corpus-
3024 # wide number that would be true of the next person's copy — the row
3025 # fills in the moment it is opened, which is the honest answer.
3026 ),
3027 # DATA-63: one dataset per reading regime, each holding every part.
3028 **{
3029 ONESTOP_REGIME_CHOICES[regime]: _onestop_regime_entry(regime)
3030 for regime in ONESTOP_REGIME_CHOICES
3031 },
3032}
3035#: The same presentation metadata for the sources that are **not** registry
3036#: entries — the packaged demo, the synthetic trial, the authoring canvas — so
3037#: the Data page can answer the same questions about every dataset.
3038#: Uploads are absent on purpose: nothing here knows anything about them that
3039#: their own row does not already show.
3040_BUILTIN_DATASET_ABOUT: dict[str, dict] = {
3041 DEMO_CHOICE: dict(
3042 language="English (L1)",
3043 # Regenerated by `python -m scanpath_studio.update_sample_data`.
3044 description="A small part of OneStop Eye Movements — native English "
3045 "speakers reading Guardian articles — bundled so the app opens with "
3046 "real data.",
3047 link="https://github.com/lacclab/OneStop-Eye-Movements",
3048 # DATA-36: this corpus ships *inside* the package, so its figures are
3049 # checked against the files themselves — the DATA-36 tests recount
3050 # them, so regenerating the subset fails a test rather than quietly
3051 # leaving a stale number in the table.
3052 published_counts={
3053 "Participants": 2,
3054 "Texts": 12,
3055 "Trials": 24,
3056 "Words": 2614,
3057 "Fixations": 3209,
3058 "Gaze points": 2233,
3059 },
3060 published_counts_source=(
3061 "Counted from the files bundled with this release of the package."
3062 ),
3063 # UX-177: OneStop publishes no raw samples, so the raw-gaze layer is
3064 # made up — which changes how that layer is read, so it is said.
3065 # VIZ-50: the flag says it at the figure too, and in its exports
3066 # (`synthesized_raw_gaze_note`), in this same sentence.
3067 reading_note="The raw-gaze samples are synthesized: OneStop publishes "
3068 "no raw gaze.",
3069 raw_gaze_synthesized=True,
3070 ),
3071 ONESTOP_CHOICE: dict(
3072 language="English (L1)",
3073 description="OneStop Eye Movements — native English speakers reading "
3074 "Guardian articles — read from the lab export at `$ONESTOP_DATA_DIR`.",
3075 link="https://github.com/lacclab/OneStop-Eye-Movements",
3076 # DATA-36: seeded from the *public* release, because that is the only
3077 # figure that can be known before the export on this machine is read.
3078 # A lab export is a superset — it carries cohorts the public release
3079 # does not — so this row is the one where loading more than was
3080 # published is expected rather than alarming; the ⚠️ is then saying
3081 # "your export is bigger than the public corpus", which is true.
3082 published_counts={"Participants": 360},
3083 published_counts_source=(
3084 "The public release's own figure (Berzak et al. 2025). A lab export "
3085 "can hold more — the L2 cohort is not in the public release — so "
3086 "treat this as a floor."
3087 ),
3088 ),
3089 MANUAL_SAMPLE_CHOICE: dict(
3090 language="English",
3091 description="A scanpath drawn by hand over a short English text, "
3092 "yours to edit.",
3093 ),
3094 SYNTHETIC_CHOICE: dict(
3095 language="English",
3096 description="A hand-built six-word English trial, for checking what a "
3097 "plot option does.",
3098 published_counts={
3099 "Participants": 1,
3100 "Texts": 1,
3101 "Trials": 1,
3102 "Words": 6,
3103 "Fixations": 9,
3104 },
3105 published_counts_source=(
3106 "The fixture's own specification — six words on two lines, nine "
3107 "fixations, one of them out of text (`synthetic.py`)."
3108 ),
3109 ),
3110 AUTHOR_CHOICE: dict(
3111 description="Type a text and place fixations on it yourself, for "
3112 "figures that illustrate a pattern rather than report a recording.",
3113 ),
3114}
3117def dataset_about(token: str, registry: dict | None = None) -> dict:
3118 """What the dataset table knows about one row beyond its counts.
3120 One lookup for both halves of the catalogue — a public corpus' registry
3121 entry and the packaged sources' table above — so neither the table's row nor
3122 the open dataset's section has to care which kind of dataset it is.
3123 ``language`` feeds the table's filter; ``description``, ``link`` and
3124 ``reading_note`` are the lines under *What's in the dataset* (UX-177). Returns ``{}`` for an upload, which is the
3125 honest answer: nothing here knows anything about it that its own row does
3126 not already show.
3127 """
3128 spec = (registry if registry is not None else public_dataset_registry()).get(token)
3129 if spec:
3130 # `published_counts` is a dict living in the registry, so it is copied
3131 # on the way out — everything else here is an immutable string, and a
3132 # caller that edited this one in place would be editing the catalogue.
3133 return {
3134 key: dict(spec[key]) if key == "published_counts" else spec[key]
3135 for key in (
3136 "language",
3137 "description",
3138 "link",
3139 "reading_note",
3140 # DATA-36: the published figures ride this same lookup rather
3141 # than a second one, so a public corpus, a packaged source and a
3142 # prepared benchmark corpus all answer for themselves the same
3143 # way — and an upload answers `{}`.
3144 "published_counts",
3145 "published_counts_source",
3146 )
3147 if spec.get(key)
3148 }
3149 about = dict(_BUILTIN_DATASET_ABOUT.get(token) or {})
3150 if "published_counts" in about:
3151 about["published_counts"] = dict(about["published_counts"])
3152 return about
3155def synthesized_raw_gaze_note(token: str | None) -> str:
3156 """The catalogue's sentence for a dataset whose raw gaze is made up, else ``""``.
3158 VIZ-50: the 🗂️ Data page says the demo's samples are synthesized, but the
3159 🔵 Raw gaze layer is switched on from Scanpath, where that page is out of
3160 sight — so the plot repeats the note while the layer is drawn, and the
3161 Share → File settings and the bundle's ``plot_config.json`` record it. One
3162 flag (``raw_gaze_synthesized``) and one sentence (``reading_note``), read
3163 from the packaged sources' table directly: no public corpus or upload sets
3164 it, so the registry is never built to answer.
3165 """
3166 about = _BUILTIN_DATASET_ABOUT.get(str(token or "")) or {}
3167 if not about.get("raw_gaze_synthesized"):
3168 return ""
3169 return str(about.get("reading_note") or "")
3172def _benchmark_registry_entries() -> dict:
3173 """One registry entry per benchmark corpus the user added (R36, DATA-55).
3175 Built from those corpora's manifest entries (`added_benchmark_datasets`),
3176 so it varies at runtime — which is why the registry as a whole had to become
3177 a function. Each entry has the same shape as the static built-ins above
3178 (`short` / `language` / `size` / `description` / `link` / `monitor`) and is
3179 presented identically: one 🌐 entry in the flat picker, nothing nested.
3181 ``monitor`` is **omitted** when the manifest's ``monitor_source`` is
3182 ``default`` — that value is `eyegenbench_geometry.py`'s generic guess for a
3183 corpus that documents no screen, and declaring it here would make the canvas
3184 snap to it as though it were measured (the registry has no "declared but not
3185 authoritative" tier — a declared monitor *is* the authoritative one). Without
3186 it the canvas falls back to data extents, which is the honest answer. The
3187 condition itself is `eyegenbench.declared_monitor`, which the CLI reads too.
3188 """
3189 from scanpath_studio.eyegenbench import declared_monitor, entry_name
3191 entries: dict = {}
3192 for entry in added_benchmark_datasets():
3193 # `entry_name` owns the "a row with no usable name is skipped" rule (N5).
3194 if not (name := entry_name(entry)):
3195 continue
3196 short = _benchmark_short_name(name)
3197 spec = dict(
3198 loader=partial(_load_benchmark_source, dataset=name),
3199 # BUG-113. No download: a bundle is prepared by a script.
3200 files_present=partial(_benchmark_files_present, name),
3201 short=short,
3202 language=language_display(entry.get("language")),
3203 size=_benchmark_size_caption(entry),
3204 description=_benchmark_description(entry, harmonised_overlap=short != name),
3205 # R34's badge, resolved once here rather than only inside the loader,
3206 # so the Data page can show it without re-reading the manifest.
3207 reading_note=geometry_badge(entry),
3208 link="https://github.com/EyeBench/EyeGenBench",
3209 # DATA-36: the manifest already counts each corpus, so a prepared
3210 # row arrives with its figures — the same numbers `size` renders as
3211 # a caption, one column each.
3212 published_counts=benchmark_published_counts(entry),
3213 published_counts_source=(
3214 "The prepared bundle's own manifest. Texts counts *distinct* "
3215 "texts, so a corpus with repeated readings publishes fewer than "
3216 "the app counts reading instances."
3217 ),
3218 # Marks the entry as coming from a prepared bundle, and names the
3219 # corpus inside it. `compare_source` dispatches on this rather than
3220 # sniffing the label, and it is the natural slug for Task 12's wire
3221 # format.
3222 benchmark_dataset=name,
3223 )
3224 # The one screen-honesty rule, shared with the CLI (`cli.render
3225 # --eyegenbench`) so the same corpus can't render at an invented
3226 # 1920×1080 on one surface and at data extents on the other — I3.
3227 if monitor := declared_monitor(entry):
3228 spec["monitor"] = monitor
3229 entries[benchmark_corpus_label(name)] = spec
3230 return entries
3233def public_dataset_registry() -> dict:
3234 """Every public corpus on offer: the static built-ins ∪ the added corpora.
3236 `PUBLIC_DATASET_REGISTRY` stays the literal home of the three built-ins, whose
3237 entries are fixed at import time. The benchmark corpora a user adds can't be,
3238 so they are composed in here and every consumer calls this instead of reading
3239 the dict. Nothing is discovered: a corpus is here only because someone added
3240 it (DATA-55; the flow that adds one is DATA-56).
3242 DATA-54 and DATA-55 hold MultiplEYE and the harmonised benchmark corpora back
3243 for the beta unless ``SCANPATH_EXPERIMENTAL`` is on. Gating here, the one
3244 place every consumer reads, is what hides them from the picker, the 🗂️ Data
3245 page, Compare's second dataset and share links at once.
3246 """
3247 registry = dict(PUBLIC_DATASET_REGISTRY)
3248 if not multipleye_enabled():
3249 registry.pop(MULTIPLEYE_PUBLIC_CHOICE, None)
3250 if benchmark_corpora_enabled():
3251 registry.update(_benchmark_registry_entries())
3252 return registry
3255def _load_public_dataset(
3256 description_host=None, options_host=None, location_host=None
3257) -> tuple[pd.DataFrame, pd.DataFrame]:
3258 """Dispatch for a "Public datasets" source.
3260 The corpus is chosen in the flat source picker (DATA-9) and rides
3261 ``public_dataset_choice``. The selected corpus' compact language · size
3262 caption and home link render into ``description_host``, under the editable
3263 description (UX-174 r2); the loader's source options + data-location controls
3264 render into ``options_host`` / ``location_host`` (the DATA-9 ordered group).
3265 Returns raw, pre-normalization frames.
3266 """
3267 registry = public_dataset_registry()
3268 chosen = st.session_state.get("public_dataset_choice")
3269 if chosen not in registry:
3270 chosen = next(iter(registry))
3271 st.session_state["public_dataset_choice"] = chosen
3272 spec = registry[chosen]
3273 desc = description_host if description_host is not None else st.container()
3274 facts = " · ".join(f for f in (spec.get("language"), spec.get("size")) if f)
3275 if facts:
3276 desc.caption(facts)
3277 if spec.get("link"):
3278 desc.markdown(f"[Home page ↗]({spec['link']})")
3279 return spec["loader"](options_host, location_host)
3282def _public_dataset_monitor(data_choice: str) -> tuple[int, int] | None:
3283 """The selected public corpus' real monitor size, or None.
3285 None when another data source is active, or when the selected dataset
3286 doesn't declare a monitor (canvas then defaults to data extents)."""
3287 if data_choice != PUBLIC_DATASETS_CHOICE:
3288 return None
3289 spec = public_dataset_registry().get(
3290 st.session_state.get("public_dataset_choice", "")
3291 )
3292 return spec.get("monitor") if spec else None
3295#: The recording-setup values a fresh session pins before anything declares
3296#: otherwise (`seed_canvas_state`'s `defaults`). The canvas is absent because it
3297#: is the source's own (`resolve_source_monitor`), and the DPI because it is
3298#: derived from the canvas and the physical width. EXP-19's share link reads the
3299#: same table to leave a setting off while it still equals it
3300#: (`url_state._link_defaults`).
3301SETUP_DEFAULTS = {
3302 "global_monitor_width_mm": 597.0,
3303 "global_viewing_distance_mm": 800.0,
3304 "global_base_font_size": 16,
3305 "global_stimulus_font_pt": 12.0,
3306 "global_use_stimulus_font_pt": False,
3307}
3309#: BUG-50 — the font controls a declared stimulus typeface overwrites, and where
3310#: `seed_canvas_state` parks their pre-snap values so leaving that corpus can put
3311#: them back. All three are wire format (share link + saved config), which is why
3312#: leaking one across a source switch outlives the session that caused it.
3313_FONT_SNAP_KEYS = (
3314 "global_base_font_size",
3315 "global_font_family",
3316 "global_scale_text_to_boxes",
3317)
3318_FONT_SNAP_RESTORE_KEY = "_font_snap_restore"
3320#: VIZ-45 — the raw-gaze layer's per-dataset default follows the font snap's
3321#: shape (BUG-50): `controls.RAW_GAZE_SEEDED_FOR_KEY` records the dataset it was
3322#: last decided for, and `controls.RAW_GAZE_SNAP_RESTORE_KEY` the value it
3323#: overwrote, so leaving that dataset can put it back. Both are recovery-cache
3324#: session keys (`persistence._SESSION_KEYS`), so a relaunch onto the same
3325#: dataset does not decide again over the user's own choice.
3326_RAW_GAZE_LAYER_KEY = "global_show_raw_gaze"
3327#: `RAW_GAZE_LINK_FOR_KEY` once the link's visit is over — not None, so the
3328#: same link, still on the URL, cannot claim another dataset.
3329_RAW_GAZE_LINK_SPENT = "\x00spent"
3332def _narrowed_raw_gaze(
3333 raw_gaze: pd.DataFrame,
3334 *,
3335 participants,
3336 metadata,
3337 ranges,
3338 trial_keys,
3339 drop_unknown=None,
3340) -> pd.DataFrame:
3341 """The samples table narrowed by the trial filters that apply to it (VIZ-45).
3343 The participant filter always, the condition filters only when the caller
3344 passes them (a raw-gaze-only dataset), and the trial-metadata keys. With no
3345 filter set it is ``raw_gaze`` itself; otherwise it is built once per filter
3346 change in a `frame_cache` and the same object is handed back on every rerun
3347 after that, rather than re-masking every sample each time."""
3348 if raw_gaze is None or raw_gaze.empty:
3349 return raw_gaze
3350 if participants is None and not metadata and not ranges and trial_keys is None:
3351 return raw_gaze
3353 def _build() -> pd.DataFrame:
3354 _, narrowed = filter_trials(
3355 _EMPTY_WORDS,
3356 raw_gaze,
3357 participants=participants,
3358 metadata=metadata,
3359 ranges=ranges,
3360 drop_unknown=drop_unknown,
3361 )
3362 if trial_keys is not None:
3363 narrowed = filter_frame_to_keys(narrowed, trial_keys)
3364 return narrowed
3366 key = (
3367 frame_fingerprint(raw_gaze),
3368 hashable_key(participants),
3369 hashable_key(metadata or {}),
3370 hashable_key(ranges or {}),
3371 hashable_key(trial_keys),
3372 hashable_key(tuple(drop_unknown or ())),
3373 )
3374 return frame_cache("raw_gaze_narrowed", key, _build)
3377#: The words frame `filter_trials` is handed beside the samples — built once.
3378_EMPTY_WORDS = empty_words_frame()
3381def seed_raw_gaze_default(
3382 session,
3383 source_key: tuple,
3384 *,
3385 samples_only: bool,
3386 link_names_layer: Callable[[], bool] | bool = False,
3387) -> None:
3388 """Turn the 🔵 Raw gaze layer on for a dataset whose only gaze is samples.
3390 VIZ-45: a dataset with raw gaze and no fixations opened as a blank plot,
3391 because the layer defaults off — the only data it had was hidden. So the
3392 default is **the dataset's**, not the app's: opening a dataset that has raw
3393 gaze and no fixations (``samples_only``, decided on the unfiltered frames)
3394 turns the layer on; opening any other dataset leaves it where it was.
3396 It is decided **once per dataset** (``source_key``, `seed_canvas_state`'s
3397 key), the way the canvas and font snaps are, which is what lets an explicit
3398 choice win: switching the layer off on that dataset sticks for as long as it
3399 stays open, because nothing decides again until the dataset changes. The
3400 value it overwrote is kept and put back on the way out, so the fixation
3401 dataset opened next keeps whatever the user had there rather than inheriting
3402 the raw-gaze one's.
3404 A share link that named the layer (``link_names_layer``, a callable so it
3405 is asked only when a decision is due) is the sender's explicit choice **for
3406 the dataset it was opened on** — the first one decided while the link is on
3407 the URL (`constants.RAW_GAZE_LINK_FOR_KEY`). There the link's value stands:
3408 nothing is stashed, and a stash an earlier visit left (the recovery cache
3409 keeps it) is dropped rather than written back over the link. Another
3410 raw-gaze-only dataset the user opens afterwards gets its own default.
3412 A built-in design preset or *Reset* forgets the decision
3413 (`controls._forget_raw_gaze_default`), so they come back to the dataset's
3414 default rather than the factory one — a raw-gaze-only dataset with the layer
3415 off shows no gaze whatever the preset's name. A saved design does not: it is
3416 the user's own record, its raw-gaze switch included.
3417 """
3418 token = "\x1f".join(str(part) for part in source_key)
3419 if session.get(RAW_GAZE_SEEDED_FOR_KEY) == token:
3420 return
3421 link_for = session.get(RAW_GAZE_LINK_FOR_KEY)
3422 if link_for is None and (
3423 link_names_layer() if callable(link_names_layer) else link_names_layer
3424 ):
3425 link_for = session[RAW_GAZE_LINK_FOR_KEY] = token
3426 elif link_for is not None and link_for != token:
3427 # A link is one visit: the first other dataset decided spends it, so
3428 # coming back to the linked dataset later is an ordinary visit — its
3429 # value would otherwise drop the stash the dataset in between made.
3430 link_for = session[RAW_GAZE_LINK_FOR_KEY] = _RAW_GAZE_LINK_SPENT
3431 from_link = link_for == token
3432 if from_link:
3433 # The link's value overwrote nothing, and a stash from before the link
3434 # must not overwrite it either.
3435 session.pop(RAW_GAZE_SNAP_RESTORE_KEY, None)
3436 elif samples_only:
3437 if RAW_GAZE_SNAP_RESTORE_KEY not in session:
3438 # Stashed on the first of a run of raw-gaze datasets only, so
3439 # raw-gaze → raw-gaze → fixations restores the pre-raw-gaze value.
3440 # `None` = absent: leaving restores the factory default.
3441 session[RAW_GAZE_SNAP_RESTORE_KEY] = {
3442 "value": session.get(_RAW_GAZE_LAYER_KEY)
3443 }
3444 session[_RAW_GAZE_LAYER_KEY] = True
3445 else:
3446 stashed = session.get(RAW_GAZE_SNAP_RESTORE_KEY)
3447 if isinstance(stashed, dict):
3448 session.pop(RAW_GAZE_SNAP_RESTORE_KEY, None)
3449 prior = stashed.get("value")
3450 if prior is None:
3451 session.pop(_RAW_GAZE_LAYER_KEY, None)
3452 else:
3453 session[_RAW_GAZE_LAYER_KEY] = bool(prior)
3454 session[RAW_GAZE_SEEDED_FOR_KEY] = token
3457def _dataset_font(words: pd.DataFrame) -> tuple[float | None, str | None]:
3458 """The stimulus typeface ``(font_px, css_family)`` a dataset declares, or
3459 ``(None, None)``.
3461 MultiplEYE stamps ``stimulus_font_px`` / ``stimulus_font_family`` (the real
3462 ``FONT_SIZE`` + font from its stimulus config) onto every word; the app snaps
3463 its font controls to them so the reading text matches the stimulus exactly."""
3464 if words is None or words.empty or "stimulus_font_px" not in words.columns:
3465 return None, None
3466 px = pd.to_numeric(words["stimulus_font_px"], errors="coerce").dropna()
3467 if px.empty:
3468 return None, None
3469 family = None
3470 if "stimulus_font_family" in words.columns:
3471 fams = words["stimulus_font_family"].dropna().astype(str)
3472 fams = fams[fams.str.strip() != ""]
3473 family = fams.iloc[0] if not fams.empty else None
3474 return float(px.iloc[0]), family
3477def _stimulus_font_install_hint(css_family: str | None) -> tuple[str, str] | None:
3478 """``(primary font name, download URL)`` for a stimulus font's CSS stack.
3480 The overlaid reading text only matches the stimulus image when the exact
3481 experiment font is installed (we don't bundle it) — the browser otherwise
3482 falls back per-script, so CJK lands but the half-width Latin in a CJK font
3483 drifts (URLs/digits render too wide). Returns the human-readable family name
3484 (first quoted entry of the stack) + a best-effort download link, or None when
3485 the stack names no specific (quoted) family — a bare CSS generic like
3486 ``monospace`` has nothing to install."""
3487 if not css_family:
3488 return None
3489 match = re.search(r"'([^']+)'", css_family)
3490 if match is None:
3491 return None
3492 name = match.group(1)
3493 # Best-effort source: the experiment fonts are from Google's Noto project.
3494 url = (
3495 "https://github.com/notofonts/noto-cjk"
3496 if "cjk" in name.lower() or "noto" in name.lower()
3497 else f"https://fonts.google.com/?query={name.replace(' ', '+')}"
3498 )
3499 return name, url
3502@st.cache_data(show_spinner=False)
3503def _cached_words_join_nothing(
3504 _words: pd.DataFrame, _fixations: pd.DataFrame, cache_key
3505) -> bool:
3506 """Whether a loaded words table shares no (participant, trial) with the
3507 fixations (BUG-32) — memoized, since it dedups both whole frames."""
3508 if _fixations.empty:
3509 return False
3510 return not trial_keys(_words) & trial_keys(_fixations)
3513#: BUG-32 — said once per page, in the notices strip, while it holds.
3514WORDS_JOIN_NOTHING_WARNING = (
3515 f"{ICONS['warning']} **No fixation has word boxes.** A Words table was loaded, but none "
3516 "of its participant + trial pairs is in the fixations, so every trial draws "
3517 "without its text or its word-level measures. The usual cause is a **Trial "
3518 "ID** or **Participant ID** mapping that names different trials in the two "
3519 "tables — for instance one carried over from another dataset with the same "
3520 f"columns. Check those two rows under {ICONS['view_data']} **Data Management → "
3521 "Edit dataset**."
3522)
3525@st.cache_data(show_spinner=False)
3526def _cached_trial_identity_report(
3527 _words: pd.DataFrame, _fixations: pd.DataFrame, cache_key, sample_trials=None
3528) -> dict:
3529 """VAL-7's diagnosis, memoized on the two frames' fingerprints.
3531 It groups the whole corpus by trial several times over, so it must not run
3532 on every rerun — but it also must not be skipped, since the failure it
3533 catches is invisible in the figure. PERF-6: on a corpus larger than
3534 ``sample_trials`` it screens a deterministic sample instead (4.23 s → 1.15 s
3535 on full OneStop); the 🗂️ Data page's *Check every trial* button asks for the
3536 census by passing ``None``, and lands on its own cache entry.
3537 """
3538 # UX-166: only a miss gets here — the report is what shows the gated
3539 # dataset card for the census's seconds.
3540 progress.report()
3541 return diagnose_trial_identity(_words, _fixations, sample_trials=sample_trials)
3544# UX-166: the dataset card lists this step.
3545@st.cache_data(show_spinner=False)
3546def _cached_multipleye_server_bundle(
3547 participant: str | None = None,
3548) -> tuple[pd.DataFrame, pd.DataFrame]:
3549 progress.report() # a miss: real work, so the gated dataset card may show
3550 return stamp_source(load_multipleye_server_bundle(participant))
3553def load_words_and_fixations(
3554 data_choice: str,
3555 participant: str | None = None,
3556 *,
3557 description_host=None,
3558 options_host=None,
3559 location_host=None,
3560) -> tuple[pd.DataFrame, pd.DataFrame]:
3561 """Load raw word + fixation frames for the **non-upload** data sources.
3563 The Upload source is handled separately by the setup wizard
3564 (``_render_data_setup``), which groups each table's upload box with its
3565 mapping; this covers the bundled demo, synthetic trial, public datasets, and
3566 the OneStop server bundle.
3568 ``description_host`` / ``options_host`` / ``location_host`` are the DATA-9
3569 sub-slots a public dataset's caption / source options / data-location
3570 controls render into (ignored by the other sources).
3572 Args:
3573 data_choice: ``DEMO_CHOICE`` ("Bundled Demo") / ``SYNTHETIC_CHOICE`` /
3574 ``PUBLIC_DATASETS_CHOICE`` / ``ONESTOP_CHOICE`` /
3575 ``MULTIPLEYE_BUNDLE_CHOICE``. The Upload source and stored uploaded
3576 datasets are handled by ``main`` directly, not here.
3577 participant: Lowercased participant_id from the URL deep link. When set
3578 AND `data_choice` is ``ONESTOP_CHOICE`` / ``MULTIPLEYE_BUNDLE_CHOICE``,
3579 the loader fast-paths to just that pid's shard/session — sub-second
3580 instead of loading the whole corpus. Ignored for the other sources.
3582 Returns:
3583 Tuple of (words_df, fixations_df) as raw DataFrames before normalization.
3584 """
3585 if data_choice == SYNTHETIC_CHOICE:
3586 from scanpath_studio.synthetic import load_synthetic_data
3588 return load_synthetic_data()
3589 if data_choice == PUBLIC_DATASETS_CHOICE:
3590 return _load_public_dataset(description_host, options_host, location_host)
3591 # The Upload source is handled separately by the setup wizard
3592 # (`_render_data_setup`), which renders each table's upload + mapping; see main().
3593 if data_choice == ONESTOP_CHOICE:
3594 words, fixations = load_onestop_server_bundle(participant=participant)
3595 if words.empty or fixations.empty:
3596 _note_dataset_unavailable(
3597 label="OneStop server bundle",
3598 reason=(
3599 "`$ONESTOP_DATA_DIR` isn't set."
3600 if onestop_data_dir() is None
3601 else "the export files aren't in `$ONESTOP_DATA_DIR`."
3602 ),
3603 action="Point the `ONESTOP_DATA_DIR` environment variable at a "
3604 "OneStop export folder and restart the app, or open a **OneStop · "
3605 "…** dataset to download the reports instead.",
3606 root=str(onestop_data_dir() or ""),
3607 )
3608 return load_sample_data()
3609 return words, fixations
3610 if data_choice == MULTIPLEYE_BUNDLE_CHOICE:
3611 try:
3612 words, fixations = _cached_multipleye_server_bundle(participant=participant)
3613 except (FileNotFoundError, ValueError, OSError) as exc:
3614 st.error(f"Couldn't load the MultiplEYE bundle: {exc}")
3615 st.stop()
3616 if words.empty or fixations.empty:
3617 _note_dataset_unavailable(
3618 label="MultiplEYE bundle",
3619 reason="its session folders weren't found.",
3620 action="Point `MULTIPLEYE_DATA_DIR` at a MultiplEYE session set "
3621 "and restart the app.",
3622 root=str(multipleye_bundle_dir() or ""),
3623 )
3624 return load_sample_data()
3625 return words, fixations
3626 return load_sample_data()
3629def _schema_key(schema: dict | None) -> tuple | None:
3630 """Hashable, stable representation of a column-mapping schema dict.
3632 Values may be strings, ``None``, or a list of column names (composite trial
3633 id). Used as part of the normalization cache key so an override that changes
3634 the mapping (without changing the raw frame) correctly busts the cache.
3635 """
3636 if schema is None:
3637 return None
3638 return tuple(
3639 (k, tuple(v) if isinstance(v, list) else v) for k, v in sorted(schema.items())
3640 )
3643#: How the last normalized pair's stimulus-level AOI table attached to its
3644#: readings (a ``data.StimulusJoin``, or ``None``), written by `_normalize_pair`
3645#: for the add-dataset wizard (DATA-49). Scratch state, not wire format.
3646STIMULUS_JOIN_KEY = "_stimulus_join"
3648#: DATA-66: the columns the last `_normalize_pair` changed the values of
3649#: (``data.Rewrite``s), which `_stash_active_mapping` marks converted in the
3650#: column-name map. Scratch state, cleared with the map each load.
3651HARMONIZE_REWRITES_KEY = "_harmonize_rewrites"
3654def _normalize_pair_uncached(
3655 _words_df: pd.DataFrame,
3656 _word_schema: dict | None,
3657 _fixations_df: pd.DataFrame,
3658 _fix_schema: dict | None,
3659 cache_key,
3660 _keep_words: set | None = None,
3661 _keep_fix: set | None = None,
3662) -> tuple[pd.DataFrame, pd.DataFrame, StimulusJoin | None, tuple]:
3663 """Pure normalize + harmonize, cached on a cheap fingerprint of the inputs.
3665 Also returns how a stimulus-level AOI table attached to the readings
3666 (the one ``data.harmonize_frames_with_join`` made, DATA-49) — ``None`` for a per-reader one —
3667 which ``_normalize_pair`` publishes for the add-dataset wizard to state.
3669 ``cache_key`` carries a ``frame_fingerprint`` + schema signature + the
3670 keep-column selection, so a trial change (which re-runs the script but feeds
3671 byte-identical raw frames) hits the cache and skips re-normalizing the whole
3672 corpus, while changing the kept columns correctly busts it.
3674 PERF-6: deliberately **not** ``@st.cache_data``. That would store a copy of
3675 its own and hand out another on every hit, so the frames would sit in memory
3676 twice — measured at ~1.2 GB of avoidable resident memory at OneStop scale,
3677 against the 0.69 s per rerun the copy was costing. `frame_cache` keeps
3678 exactly one, which is why `clear_computation_cache` clears it too.
3679 """
3680 # UX-37: logged because a cache *miss* is exactly what a "why was that slow?"
3681 # question is about, and a hit is silent — the line only appears when the
3682 # work actually ran.
3683 with timed(
3684 "normalize + harmonize (cache miss)",
3685 word_rows=len(_words_df),
3686 fixation_rows=len(_fixations_df),
3687 ):
3688 # UX-166 (T5-3): three reportable parts, so a cancel checkpoint exists
3689 # partway through instead of only at the very end — on a real corpus
3690 # this stage alone is the ~20 s wait the spinner above describes.
3691 progress.report(0, 3, detail="words")
3692 words_norm = (
3693 normalize_words(_words_df, _word_schema, keep_columns=_keep_words)
3694 if _word_schema is not None
3695 else empty_words_frame()
3696 )
3697 progress.report(1, 3, detail="fixations")
3698 fixations_norm = (
3699 normalize_fixations(_fixations_df, _fix_schema, keep_columns=_keep_fix)
3700 if _fix_schema is not None
3701 else empty_fixations_frame()
3702 )
3703 progress.report(2, 3, detail="matching tables")
3704 # The join the fixups actually made, after BUG-59's zero padding — never
3705 # a plan of the frames before it, which can disagree. DATA-66: and the
3706 # columns whose values they changed, for the column-name map.
3707 words_norm, fixations_norm, join, rewrites = harmonize_frames_reporting(
3708 words_norm, fixations_norm
3709 )
3710 progress.report(3, 3)
3711 return words_norm, fixations_norm, join, rewrites
3714def _normalize_pair(
3715 words_df: pd.DataFrame,
3716 word_schema: dict | None,
3717 fixations_df: pd.DataFrame,
3718 fix_schema: dict | None,
3719 keep_words: set | None = None,
3720 keep_fix: set | None = None,
3721) -> tuple[pd.DataFrame, pd.DataFrame]:
3722 """Normalize a *validated* (words, fixations) pair to canonical columns and
3723 run the cross-frame fixups (``harmonize_frames``).
3725 A ``None`` schema means that table is absent (single-report dataset) → a
3726 canonical empty frame. Records the composite-trial component columns (when
3727 the trial id is built from several columns) so the trial picker can offer one
3728 cascading selector per component. Shared by the upload and non-upload paths.
3730 The heavy normalization is ``_normalize_pair_uncached``, cached through
3731 ``frame_cache`` on a fingerprint key (PERF-6) so it doesn't re-run on every
3732 rerun (e.g. selecting a different trial); only the lightweight session-state
3733 bookkeeping below runs each time.
3734 """
3735 trial_mapping = (word_schema or fix_schema)["trial"]
3736 trial_cols = trial_mapping_columns(trial_mapping)
3737 st.session_state["_composite_trial_columns"] = (
3738 trial_cols if len(trial_cols) > 1 else None
3739 )
3740 cache_key = (
3741 frame_fingerprint(words_df),
3742 _schema_key(word_schema),
3743 frame_fingerprint(fixations_df),
3744 _schema_key(fix_schema),
3745 tuple(sorted(keep_words)) if keep_words is not None else None,
3746 tuple(sorted(keep_fix)) if keep_fix is not None else None,
3747 )
3748 # PERF-6: the normalized frames are cached *only* here. `st.cache_data`
3749 # hands back a deep copy on every hit — ~0.69 s and ~0.7 GB of churn per
3750 # rerun at OneStop scale, for frames the app already has and never writes to
3751 # (audited by tests/test_frame_immutability.py) — and keeping both caches
3752 # would hold two copies of the corpus at rest, which is the worse of the two
3753 # costs. `frame_cache` keeps one and returns the object itself.
3754 # PERF-6: the spinner says how much data is being normalized, because on a
3755 # real corpus this is a ~20 s wait and "Normalizing data…" gives no sense of
3756 # whether that is expected. UX-166 moved it *outside* the cache: a spinner's
3757 # exit is a yield point, and inside `build` an abandoned run raised there
3758 # and threw the finished normalization away. `st.spinner` shows only after
3759 # 0.5 s, so a cache hit still never flashes it, and `loading.spinner` stays
3760 # silent under the dataset card, which lists normalization as a step.
3761 with loading.spinner(
3762 f"Mapping {len(words_df):,} word rows and {len(fixations_df):,} fixations…"
3763 ):
3764 words_norm, fixations_norm, join, rewrites = frame_cache(
3765 "normalized_pair",
3766 cache_key,
3767 lambda: _normalize_pair_uncached(
3768 words_df,
3769 word_schema,
3770 fixations_df,
3771 fix_schema,
3772 cache_key,
3773 _keep_words=keep_words,
3774 _keep_fix=keep_fix,
3775 ),
3776 # PERF-18: the dataset before this one stays normalized, so
3777 # switching back to it is instant.
3778 keep=2,
3779 )
3780 # DATA-49: which key a stimulus-level AOI table joined through, for the
3781 # add-dataset wizard to say — bookkeeping like `_composite_trial_columns`
3782 # above, written on a cache hit too so it always describes this pair.
3783 st.session_state[STIMULUS_JOIN_KEY] = join
3784 st.session_state[HARMONIZE_REWRITES_KEY] = rewrites
3785 return words_norm, fixations_norm
3788def _reset_active_mapping() -> None:
3789 """Clear the stashed column mapping at the start of each data load, so a new
3790 source doesn't inherit the previous one's mapping in the Data Inspection tab."""
3791 st.session_state["_active_column_mapping"] = {}
3792 st.session_state[ACTIVE_COLUMN_NAMES_KEY] = {}
3793 st.session_state.pop(HARMONIZE_REWRITES_KEY, None)
3796def _stash_active_mapping(
3797 table: str,
3798 schema: dict | None,
3799 columns: Iterable[str] | None = None,
3800 *,
3801 keep_columns: Iterable[str] | None = None,
3802 names: ColumnNames | None = None,
3803) -> None:
3804 """Record the schema (field → source column) actually used for ``table`` so
3805 ``tabs.render_data_inspection_tab`` can show how columns were mapped. ``table``
3806 is one of ``"words" / "fixations" / "raw_gaze"``.
3808 DATA-66: also the column-name map the schema implies — ``names`` when the
3809 caller already holds one (a stored upload), else built from the raw table's
3810 ``columns``. Neither: the table's map is dropped, never left stale."""
3811 mapping = st.session_state.setdefault("_active_column_mapping", {})
3812 mapping[table] = dict(schema) if schema else None
3813 stash = st.session_state.setdefault(ACTIVE_COLUMN_NAMES_KEY, {})
3814 if names is None and schema and columns is not None:
3815 names = from_schema(
3816 table, schema, columns, keep_columns=keep_columns
3817 ).with_rewrites(table, st.session_state.get(HARMONIZE_REWRITES_KEY))
3818 if names is None:
3819 stash.pop(table, None)
3820 else:
3821 stash[table] = names.to_payload()
3824def active_column_names(table: str) -> ColumnNames:
3825 """The open dataset's column-name map for ``table`` (DATA-66)."""
3826 return active_names(st.session_state, table)
3829#: Lead of the ``problems`` entry a **rejected** mapping produces, as opposed to
3830#: an **incomplete** one. ``_render_unmapped_view`` branches on it to say the
3831#: right thing, so keep the two in step.
3832MAPPING_FAILURE_LEAD = "This column mapping doesn't work with this data"
3835def mapping_failure_problem(exc: Exception) -> str:
3836 """Turn a normalization failure into one more recovery ``problems`` entry.
3838 Everything the normalize → harmonize pipeline raises is a statement about
3839 the column mapping in force — a ``multipart`` identity rule (screen id in
3840 only one report, orphan screens, a conflicting canvas), a non-numeric
3841 coordinate column, a duplicated trial key. None of them is a reason to stop
3842 rendering, and letting one propagate is actively a trap: the panels that
3843 would fix the mapping are written *during* the run that dies, and the Data
3844 page they live on is hidden while another view is active — so the user is
3845 left with a traceback and nothing to click, which is how the app used to
3846 wedge on a mapping it had auto-detected itself.
3848 The exception is logged with its traceback (🐛 Debug panel + terminal) and
3849 handed back as a string, so the existing incomplete-mapping recovery path —
3850 raw tables, still-editable mapping panels, off-page signpost — carries it.
3851 """
3852 logging.getLogger("scanpath_studio").exception(
3853 "Normalizing with the current column mapping failed."
3854 )
3855 if not isinstance(exc, ValueError):
3856 # A KeyError or pandas TypeError reads as a bare repr (``'x'``); the
3857 # traceback is in the log above.
3858 return (
3859 f"{MAPPING_FAILURE_LEAD}: a column it names holds values the app "
3860 "can't read (details in Help → Debug)"
3861 )
3862 return f"{MAPPING_FAILURE_LEAD}: {exc}"
3865def reset_column_mapping() -> None:
3866 """Drop every ``col_map_*`` key, so the mapping falls back to auto-detection.
3868 Used both when the monitor-defining source changes (a mapping is keyed to
3869 the columns it was made for) and as the escape hatch under a rejected
3870 mapping. Safe from an ``on_click`` callback: it runs before the script that
3871 re-creates the widgets.
3872 """
3873 for key in [
3874 k
3875 for k in list(st.session_state)
3876 if isinstance(k, str) and k.startswith(COLUMN_MAPPING_PREFIX)
3877 ]:
3878 del st.session_state[key]
3881#: A built-in source (the demo, a public corpus) maps its columns with the
3882#: `col_map_*` panels themselves, which apply as they change. While ✏️ Edit
3883#: dataset is open they are a draft instead, like an upload's editor: the
3884#: dataset keeps the mapping it had when the editor opened (held here, with the
3885#: `col_map_*` keys that produced it) until ✅ Save changes adopts the draft, and
3886#: ✕ Cancel puts the keys back.
3887BUILTIN_MAPPING_HELD_KEY = "_builtin_mapping_held"
3888#: The panels' current picks, ``{"words": schema, "fixations": schema}``,
3889#: written by `prepare_data` on every run that draws them.
3890BUILTIN_MAPPING_PENDING_KEY = "_builtin_mapping_pending"
3891#: Set by ✅ Save changes for the success line on the screen it returns to.
3892BUILTIN_MAPPING_SAVED_KEY = "_builtin_mapping_saved"
3893#: ✕ Cancel's restore, parked for the next run to apply before the panels draw:
3894#: the Leave confirmation is a dialog, whose click runs inside the dialog's own
3895#: rerun rather than ahead of the page's widgets.
3896BUILTIN_MAPPING_RESTORE_KEY = "_builtin_mapping_restore"
3897#: The panels a built-in source draws (`prepare_data`). Only their field
3898#: values are held: not the per-cell confirm buttons (`*_cell_confirm`, whose
3899#: value Streamlit refuses to have set), the add wizard's `*_upload` files or
3900#: its stashed `*_header` — the scaffolding `tabs._EDITOR_KEY_NOISE` names.
3901_BUILTIN_PANEL_PREFIXES = ("col_map_words_", "col_map_fix_")
3904def _is_builtin_panel_key(key) -> bool:
3905 return (
3906 isinstance(key, str)
3907 and key.startswith(_BUILTIN_PANEL_PREFIXES)
3908 and not any(noise in key for noise in _EDITOR_KEY_NOISE)
3909 )
3912def _mapping_signature(schemas: dict | None) -> str:
3913 """A comparable rendering of a ``{"words", "fixations"}`` mapping — lists
3914 (a composite Trial ID) come back from the widgets as new objects."""
3915 return json.dumps(schemas or {}, sort_keys=True, default=str)
3918def held_builtin_mapping(source_key) -> dict | None:
3919 """The mapping a built-in source keeps while its editor is open, or None."""
3920 held = st.session_state.get(BUILTIN_MAPPING_HELD_KEY)
3921 if not held or held.get("source") != source_key:
3922 return None
3923 return held.get("schemas")
3926def hold_builtin_mapping(source_key) -> None:
3927 """Record the mapping this run applied, and the keys behind it, as what an
3928 editor opened on the next run starts from and what its ✕ Cancel restores."""
3929 st.session_state[BUILTIN_MAPPING_HELD_KEY] = {
3930 "source": source_key,
3931 "schemas": copy.deepcopy(st.session_state.get(BUILTIN_MAPPING_PENDING_KEY)),
3932 "keys": {
3933 key: copy.deepcopy(value)
3934 for key, value in st.session_state.items()
3935 if _is_builtin_panel_key(key)
3936 },
3937 }
3940def builtin_mapping_is_dirty(source_key) -> bool:
3941 """Whether the open editor's mapping panels differ from the held mapping."""
3942 held = held_builtin_mapping(source_key)
3943 if held is None:
3944 return False
3945 return _mapping_signature(
3946 st.session_state.get(BUILTIN_MAPPING_PENDING_KEY)
3947 ) != _mapping_signature(held)
3950def _discard_builtin_mapping_edit() -> None:
3951 """Ask the next run to put the panels back as they were when the editor
3952 opened (`restore_builtin_mapping`)."""
3953 held = st.session_state.pop(BUILTIN_MAPPING_HELD_KEY, None)
3954 if held:
3955 st.session_state[BUILTIN_MAPPING_RESTORE_KEY] = held
3958def restore_builtin_mapping(source_key) -> None:
3959 """Apply a parked ✕ Cancel to ``source_key``'s panels, before they draw.
3961 A restore parked for another source is dropped: its keys describe columns
3962 this source does not have."""
3963 held = st.session_state.pop(BUILTIN_MAPPING_RESTORE_KEY, None)
3964 if not held or held.get("source") != source_key:
3965 return
3966 saved = held.get("keys") or {}
3967 for key in [k for k in list(st.session_state) if _is_builtin_panel_key(k)]:
3968 if key not in saved:
3969 del st.session_state[key]
3970 for key, value in saved.items():
3971 st.session_state[key] = value
3974def _save_builtin_mapping(mapping: bool = True) -> None:
3975 """✅ Save changes for a built-in source: adopt the draft mapping, name,
3976 description and metadata tables.
3978 A draft that leaves a required field empty is refused here, with the
3979 reasons shown above the button, rather than applied and then failing.
3980 ``mapping=False`` is a source with no mapping panels to adopt."""
3981 pending = (
3982 (st.session_state.get(BUILTIN_MAPPING_PENDING_KEY) or {}) if mapping else {}
3983 )
3984 problems: dict = {}
3985 for table_key, validate in (
3986 ("words", validate_word_schema),
3987 ("fixations", validate_fix_schema),
3988 ):
3989 schema = pending.get(table_key)
3990 if schema is not None and (found := validate(schema)):
3991 problems[table_key] = found
3992 if problems:
3993 st.session_state["_remap_problems"] = problems
3994 return
3995 # Dropped first, so closing the editor does not restore the held keys.
3996 st.session_state.pop(BUILTIN_MAPPING_HELD_KEY, None)
3997 # …and the Recording setup, before closing sweeps the form's state away.
3998 commit_builtin_setup()
3999 saved = str(st.session_state.get("data_source_choice") or "")
4000 commit_editor_staging(saved)
4001 _close_dataset_editor()
4002 st.session_state[BUILTIN_MAPPING_SAVED_KEY] = saved
4005def _render_builtin_editor_footer(host, *, mapping: bool = True) -> None:
4006 """✅ Save changes at the foot of a built-in source's ✏️ Edit dataset screen.
4008 `tabs.render_dataset_editor_footer`'s row, for a dataset with no stored
4009 entry: the same divider, the same blockers, the button in the same column.
4010 There is no ⬇️ Download setup file beside it — the corpus' own loader is the setup.
4011 ``mapping=False``: a source with no mapping panels, whose Save holds the
4012 name, description and metadata tables.
4013 """
4014 from scanpath_studio.wizard import _FOOTER_ROW_W
4016 box = host.container()
4017 box.container(key="wizard_footer_divider_edit").divider()
4018 for table_key, messages in (st.session_state.get("_remap_problems") or {}).items():
4019 label = _TABLE_LABELS.get(table_key, table_key)
4020 for message in messages:
4021 box.error(f"**{label}** — {message}", icon=ICONS["error"])
4022 row = box.container(key="wizard_footer_row_edit")
4023 _setup_col, apply_col, _rest = row.columns(
4024 _FOOTER_ROW_W, gap="small", vertical_alignment="center"
4025 )
4026 apply_col.button(
4027 f"{ICONS['confirm']} Save changes",
4028 type="primary",
4029 key="builtin_mapping_save",
4030 on_click=_save_builtin_mapping,
4031 args=(mapping,),
4032 width="stretch",
4033 help="Save the name, description, column mapping, recording setup and "
4034 "metadata tables above."
4035 if mapping
4036 else "Save the name, description and metadata tables above.",
4037 )
4040#: Label + tooltip of the off-page signpost's "known-good state" button.
4041DEMO_RESET_LABEL = f"{ICONS['demo']} Load the bundled demo"
4042DEMO_RESET_HELP = (
4043 "Switches to the bundled demo and re-detects its column mapping. Your "
4044 "uploaded datasets stay in the list."
4045)
4048def load_bundled_demo() -> None:
4049 """``on_click``: return to the bundled demo with a freshly detected mapping.
4051 The one button that reaches a known-good state from anywhere, for a session
4052 wedged on a dataset it cannot normalize. Three things together, because any
4053 two of them leave a way to stay stuck: the source switch (through the
4054 pre-widget ``_pending_source_choice`` seam — assigning the picker's value
4055 inline is reconciled away by the browser), leaving the wizard the way its
4056 own ✕ Cancel does, and dropping the column mapping — which a source *change*
4057 already does, but the wedged source is often the demo itself, and then
4058 nothing would change without this.
4059 """
4060 st.session_state["_pending_source_choice"] = DEMO_CHOICE
4061 st.session_state["_show_upload_wizard"] = False
4062 st.session_state["setup_complete"] = True
4063 st.session_state.pop(WIZARD_LEAVE_KEY, None)
4064 st.session_state.pop(WIZARD_STAY_KEY, None)
4065 reset_column_mapping()
4068def clear_computation_cache() -> None:
4069 """``on_click``: drop every ``@st.cache_data`` entry for this process.
4071 Used after deleting a dataset so derived values cannot retain its frames.
4072 It does not touch the recovery cache or live session state, except for
4073 DATA-32's remembered dataset counts, which are derived too.
4074 """
4075 st.cache_data.clear()
4076 # PERF-6: the normalized frames live in `frame_cache`, not `st.cache_data`,
4077 # so clearing only the latter would leave the deleted dataset's frames
4078 # behind — which is exactly what this function exists to prevent.
4079 clear_frame_cache()
4080 forget_dataset_counts()
4083def _apply_declared_schema(proposed: dict, declared: dict | None) -> dict:
4084 """Auto-detection, overridden by whatever the source *declares* it knows.
4086 Auto-detection guesses a mapping from column names, which is right for an
4087 upload and wrong for a corpus whose schema is a published contract. The
4088 declared mapping wins for every field it names — including a field it names
4089 as ``None``, which is a positive statement that the source has no such
4090 column and is what clears a leftover the detector would otherwise seize on.
4091 Fields the source says nothing about keep their detected value, so optional
4092 passthroughs (linguistic features, EyeLink measures) still arrive.
4093 """
4094 if not declared:
4095 return proposed
4096 return {**proposed, **declared}
4099def declared_schemas_for(data_choice: str) -> tuple[dict | None, dict | None]:
4100 """The ``(word, fix)`` schemas the selected source publishes, or ``(None, None)``.
4102 A prepared benchmark corpus has a **known** schema — the prep script wrote
4103 it — and every other surface already loads one through it
4104 (`eyegenbench.load_eyegenbench`, so `render --eyegenbench`, the headless API
4105 and Comparisons' dataset B all agree). The app was the one surface that
4106 re-guessed instead, and the guess is wrong on real bundles: the prepared
4107 frames carry the publisher's ~190 leftover columns through, so EMTeC's
4108 fixations detect `trial="TRIAL_ID"` against the words' `unique_paragraph_id`
4109 and broadcast **zero** word boxes — silently, since only the words frame
4110 ends up empty and the empty-pool guard never fires.
4112 A native corpus whose identity is a published contract declares its schema
4113 on its registry entry (``declared_schemas``): PoTeC's Trial ID is the reader
4114 *and* the text, which no column name says and detection would guess as the
4115 text alone.
4116 """
4117 if data_choice != PUBLIC_DATASETS_CHOICE:
4118 return None, None
4119 # The corpus isn't here and the demo stands in for it: its frames are the
4120 # demo's, which the corpus' schema does not describe.
4121 if st.session_state.get(_PLACEHOLDER_SHOWN_KEY):
4122 return None, None
4123 spec = public_dataset_registry().get(
4124 st.session_state.get("public_dataset_choice", "")
4125 )
4126 if spec and spec.get("declared_schemas"):
4127 word_schema, fix_schema = spec["declared_schemas"]
4128 return dict(word_schema), dict(fix_schema)
4129 if not spec or not spec.get("benchmark_dataset"):
4130 return None, None
4131 from scanpath_studio.eyegenbench import (
4132 EYEGENBENCH_FIX_SCHEMA,
4133 EYEGENBENCH_WORD_SCHEMA,
4134 )
4136 return dict(EYEGENBENCH_WORD_SCHEMA), dict(EYEGENBENCH_FIX_SCHEMA)
4139def _proposed_schema(kind: str, frame: pd.DataFrame) -> dict:
4140 """The auto-detected mapping for one raw table, worked out once per table.
4142 Detection reads values, not only names — BUG-99 rules a box column out when
4143 no cell parses as a number — so on a corpus-sized table it costs ~1.4 s
4144 (OneStop's IA report). The answer depends on nothing but the table, and
4145 `prepare_data` asks again on every rerun, so it is kept per table under the
4146 table's fingerprint. A copy is handed out: callers layer their own fields on.
4147 """
4148 propose = propose_word_schema if kind == "words" else propose_fix_schema
4149 return dict(
4150 frame_cache(
4151 f"proposed_{kind}_schema",
4152 frame_fingerprint(frame),
4153 lambda: propose(frame),
4154 # PERF-18: as the normalized pair keeps two, so going back to the
4155 # previous dataset doesn't detect its columns again.
4156 keep=2,
4157 )
4158 )
4161def prepare_data(
4162 words_df: pd.DataFrame,
4163 fixations_df: pd.DataFrame,
4164 allow_override: bool,
4165 mapping_host=None,
4166 declared_word_schema: dict | None = None,
4167 declared_fix_schema: dict | None = None,
4168 mapping_dataset: object = None,
4169 held_schemas: dict | None = None,
4170) -> tuple[pd.DataFrame, pd.DataFrame, list]:
4171 """Infer schemas and normalize incoming dataframes to canonical column names.
4173 When ``allow_override`` is True, render the mapping expanders that let the
4174 user pick the exact column names for each field (pre-filled with auto-detection).
4175 Otherwise just auto-detect.
4177 Returns ``(words_norm, fixations_norm, problems)``. ``problems`` is a list
4178 of human-readable strings; when it's non-empty the column mapping isn't
4179 usable yet (a required field is unmapped) — the normalized frames come back
4180 empty and the caller shows the raw uploaded data so the user can pick the
4181 right columns instead of the whole app halting (which used to hide the very
4182 data needed to decide the mapping).
4184 Either frame may arrive empty (single-report datasets: only an IA report,
4185 or only a fixation report) — the missing side becomes a canonical empty
4186 frame and its mapping UI is skipped. Cross-frame fixups (stimulus-level
4187 words broadcast across participants, AOI-only fixations placed at word-box
4188 centers) run at the end via ``harmonize_frames``.
4190 ``mapping_dataset`` identifies the source these tables came from, so a
4191 column pick made for another dataset — the add-dataset wizard shares these
4192 ``col_map_*`` keys — is dropped rather than inherited because the headers
4193 happen to match (BUG-32; ``controls.forget_mapping_for_other_table``).
4195 ``held_schemas`` (``{"words": …, "fixations": …}``) is the mapping the
4196 dataset keeps while ✏️ Edit dataset is open: the panels still render and
4197 their picks are published as the draft (``BUILTIN_MAPPING_PENDING_KEY``),
4198 but the frames are normalized under the held mapping until ✅ Save changes.
4199 """
4200 has_words = not words_df.empty
4201 has_fixations = not fixations_df.empty
4202 word_schema = None
4203 fix_schema = None
4204 problems: list = []
4205 pending: dict = {}
4207 if has_words:
4208 word_proposed = _apply_declared_schema(
4209 _proposed_schema("words", words_df), declared_word_schema
4210 )
4211 if allow_override:
4212 word_schema = column_mapping_ui(
4213 words_df,
4214 table_label="Words (interest areas)",
4215 state_key_prefix="col_map_words",
4216 field_specs=WORD_FIELD_SPECS,
4217 proposed=word_proposed,
4218 problems=validate_word_schema(word_proposed),
4219 container=mapping_host,
4220 # The host is the ⚙️ Configure menu popover, which nests no
4221 # expander — render the panel inline with its own bold header.
4222 use_expander=False,
4223 # Match the add-dataset screen's compact field grid instead of
4224 # stretching every mapping across a full row.
4225 columns_per_row=4,
4226 stack_labels=True,
4227 dataset=mapping_dataset,
4228 )
4229 pending["words"] = word_schema
4230 if held_schemas and held_schemas.get("words") is not None:
4231 word_schema = held_schemas["words"]
4232 else:
4233 word_schema = word_proposed
4234 word_problems = validate_word_schema(word_schema)
4235 if word_problems:
4236 problems.append("Words table: " + "; ".join(word_problems))
4238 if has_fixations:
4239 fix_proposed = _apply_declared_schema(
4240 _proposed_schema("fixations", fixations_df), declared_fix_schema
4241 )
4242 if allow_override:
4243 fix_schema = column_mapping_ui(
4244 fixations_df,
4245 table_label="Fixations",
4246 state_key_prefix="col_map_fix",
4247 field_specs=FIX_FIELD_SPECS,
4248 proposed=fix_proposed,
4249 problems=validate_fix_schema(fix_proposed),
4250 container=mapping_host,
4251 use_expander=False,
4252 columns_per_row=4,
4253 stack_labels=True,
4254 dataset=mapping_dataset,
4255 )
4256 pending["fixations"] = fix_schema
4257 if held_schemas and held_schemas.get("fixations") is not None:
4258 fix_schema = held_schemas["fixations"]
4259 else:
4260 fix_schema = fix_proposed
4261 fix_problems = validate_fix_schema(fix_schema)
4262 if fix_problems:
4263 problems.append("Fixations: " + "; ".join(fix_problems))
4265 if allow_override:
4266 st.session_state[BUILTIN_MAPPING_PENDING_KEY] = pending
4268 if problems:
4269 # Mapping not ready — let the caller surface the raw data instead of
4270 # plotting. Clear any stale composite-trial state so the picker doesn't
4271 # reference columns from a previous, valid dataset.
4272 st.session_state["_composite_trial_columns"] = None
4273 return empty_words_frame(), empty_fixations_frame(), problems
4275 # Record the mapping actually used so the Data Inspection tab can show it.
4276 _stash_active_mapping("words", word_schema if has_words else None, words_df.columns)
4277 _stash_active_mapping(
4278 "fixations", fix_schema if has_fixations else None, fixations_df.columns
4279 )
4281 try:
4282 words_norm, fixations_norm = _normalize_pair(
4283 words_df, word_schema, fixations_df, fix_schema
4284 )
4285 except Exception as exc:
4286 # A mapping the pipeline *rejects* recovers the same way as one that is
4287 # merely incomplete. See `mapping_failure_problem`.
4288 st.session_state["_composite_trial_columns"] = None
4289 return (
4290 empty_words_frame(),
4291 empty_fixations_frame(),
4292 [mapping_failure_problem(exc)],
4293 )
4294 return words_norm, fixations_norm, problems
4297# Labels of the top-level tab strip, shared by the real tabs, the
4298# unmapped-data placeholder view, and the tab-persistence script so they can't
4299# drift apart.
4300# Bulk export is no longer a top-level tab — it's folded into the Scanpath
4301# Visualization tab's "Export" subtab (see tabs._render_export_panel).
4302# The two top-level views. Scanpath is the default page; Corpus Analysis is
4303# reached via the header button (``_render_about_panel``). Data Inspection and
4304# Share are now subtabs of the Scanpath view (tabs.render_single_trial_tab),
4305# not standalone views. ``main_nav`` (session state) holds the active view.
4308def _render_raw_preview(label: str, df: pd.DataFrame) -> None:
4309 """Show one uploaded table's columns + a sample so the user can map it."""
4310 if df is None or df.empty:
4311 return
4312 st.markdown(
4313 f"#### {label} — {plural(len(df), 'row')} × {plural(df.shape[1], 'column')}"
4314 )
4315 st.caption("Columns: " + ", ".join(str(c) for c in df.columns))
4316 st.dataframe(df.head(200), width="stretch", height=320)
4319def _render_unmapped_view(
4320 raw_words_df: pd.DataFrame,
4321 raw_fixations_df: pd.DataFrame,
4322 problems: list,
4323) -> None:
4324 """Show the raw uploaded data while the column mapping isn't usable.
4326 Two different failures land here, and they need different words. A mapping
4327 that is **incomplete** asks the user to fill a field in; one the pipeline
4328 **rejected** (``MAPPING_FAILURE_LEAD``) already names what is wrong with the
4329 combination they have — so it gets the error, the reason, and a one-click
4330 way back to the auto-detected mapping, for when the offending pick came from
4331 a restored session and editing the panel field by field is a scavenger hunt.
4333 Either way the uploaded tables (unmodified) are shown below, so the user can
4334 inspect column names and values while choosing.
4335 """
4336 rejected = [p for p in problems if p.startswith(MAPPING_FAILURE_LEAD)]
4337 if rejected:
4338 for problem in rejected:
4339 st.error(problem, icon=ICONS["error"])
4340 st.caption(
4341 "Change the field it names in **2 · Data tables & column mapping** above, "
4342 "or start again from what auto-detection proposes."
4343 )
4344 st.button(
4345 f"{ICONS['undo']} Reset to the auto-detected mapping",
4346 key="reset_column_mapping",
4347 on_click=reset_column_mapping,
4348 )
4349 else:
4350 st.warning(
4351 "**Finish the column mapping to draw scanpaths.** Map what is missing "
4352 "in **2 · Data tables & column mapping** above — the raw data below "
4353 "helps you choose. "
4354 "Still needed:\n\n" + "\n".join(f"- {p}" for p in problems)
4355 )
4356 if (raw_words_df is None or raw_words_df.empty) and (
4357 raw_fixations_df is None or raw_fixations_df.empty
4358 ):
4359 st.info("No data loaded yet.")
4360 _render_raw_preview("Words (interest areas)", raw_words_df)
4361 _render_raw_preview("Fixations", raw_fixations_df)
4364def _render_dataset_load_failure(name: str, problems: list) -> None:
4365 """BUG-100: say on the Data overview that the dataset didn't load, and why.
4367 :func:`_render_unmapped_view` draws into the ✏️ Edit dataset screen, which
4368 is hidden until it is opened — so a corpus the pipeline rejected (OneStop ·
4369 Ordinary reading's orphan screens, before DATA-63) left the overview with a
4370 "Not loaded" row and no other trace. This is the overview's half: the
4371 dataset's name, the reason, and the two ways on — the editor that can fix
4372 the mapping, or back to the demo.
4373 """
4374 rejected = [p for p in problems if p.startswith(MAPPING_FAILURE_LEAD)]
4375 with st.container(border=True, key="dataset_load_failure_panel"):
4376 if rejected:
4377 for problem in rejected:
4378 reason = problem.removeprefix(MAPPING_FAILURE_LEAD).lstrip(": ")
4379 st.error(
4380 f"**{name} didn't load.** Its column mapping doesn't fit: {reason}",
4381 icon=ICONS["error"],
4382 )
4383 else:
4384 st.warning(
4385 f"**{name} isn't loaded yet** — its column mapping is "
4386 "incomplete:\n\n" + "\n".join(f"- {p}" for p in problems)
4387 )
4388 edit, demo = st.columns(2)
4389 edit.button(
4390 f"{ICONS['edit']} Edit dataset",
4391 key="dataset_load_failure_edit",
4392 on_click=_open_mapping_editor,
4393 type="primary",
4394 width="stretch",
4395 )
4396 demo.button(
4397 DEMO_RESET_LABEL,
4398 key="dataset_load_failure_demo",
4399 on_click=load_bundled_demo,
4400 width="stretch",
4401 help=DEMO_RESET_HELP,
4402 )
4405@st.cache_data(show_spinner=False)
4406def _cached_participant_ids(_words, _fixations, cache_key) -> list:
4407 """Every reader id in the dataset (DATA-20), memoized per frame pair.
4409 Underscore-prefixed frames + an explicit `frame_fingerprint` key, the house
4410 convention: this is a `.unique()` over the *unfiltered* corpus, which is
4411 hundreds of milliseconds on a full-size one.
4412 """
4413 del cache_key
4414 return metadata_mod.participant_ids(_words, _fixations)
4417def _refresh_participant_metadata(participants) -> None:
4418 """Re-report an attached participant table against the loaded readers.
4420 The table outlives a data-source switch (it is session state, like the
4421 annotations), so the join it was validated against can go stale the moment
4422 a different corpus loads. Recomputing the report — not the fields — keeps
4423 "no row for these readers" honest without asking the user to re-upload.
4424 """
4425 from scanpath_studio import metadata as md
4427 attached = st.session_state.get(md.SESSION_KEY)
4428 if attached is None:
4429 return
4430 st.session_state[md.SESSION_KEY] = md.rejoin(attached, participants)
4433def _render_offpage_setup_notice(data_view: bool) -> None:
4434 """Point at the **Data** page when setup is unfinished and we're elsewhere.
4436 DATA-26: an unfinished dataset (a wizard mid-flight, or a required column
4437 still unmapped) leaves the analysis views with nothing to draw — `main`
4438 returns before them, exactly as it did before the page existed. What it used
4439 to leave behind was a blank screen; the setup UI now lives on a page that is
4440 rendered but hidden, so say where it went and offer one click to get there.
4442 Deliberately *not* a forced `switch_to_view`: bouncing the user back every
4443 run would make the other two views unreachable until the mapping is fixed,
4444 and that is a worse trap than an empty page with a signpost.
4446 The second button is the way out that does **not** go through the page:
4447 finishing the setup is the right answer when the dataset is nearly there,
4448 but a dataset the pipeline rejects can leave the user with nothing to plot
4449 and no appetite for the mapping — and the source picker itself lives on the
4450 page they'd rather not visit. See :func:`load_bundled_demo`.
4451 """
4452 if data_view:
4453 return
4454 st.info(
4455 "**This dataset isn't set up yet**, so there's nothing to plot. "
4456 f"Finish it on the {ICONS['view_data']} **Data Management** page — or start over from the demo.",
4457 icon=ICONS["view_data"],
4458 )
4459 finish, demo = st.columns(2)
4460 finish.button(
4461 f"{ICONS['view_data']} Go to Data Management",
4462 on_click=_go_data,
4463 type="primary",
4464 width="stretch",
4465 key="offpage_go_to_setup",
4466 )
4467 demo.button(
4468 DEMO_RESET_LABEL,
4469 on_click=load_bundled_demo,
4470 width="stretch",
4471 key="offpage_load_demo",
4472 help=DEMO_RESET_HELP,
4473 )
4476# File types accepted by every upload box. ``zip`` covers single-member
4477# archives wrapping any of the others (e.g. ``data.csv.zip``). ``txt`` is the
4478# tab-separated report many exporters write (DATA-41); a text file's delimiter
4479# is read off its header line, and an ``.xls`` that is really text (EyeLink
4480# Data Viewer's "Excel" export) is read as text (BUG-55).
4481_UPLOAD_TYPES = list(UPLOAD_FILE_TYPES)
4484def _uploaded_file_key(uploaded) -> tuple:
4485 """Stable cache key for an uploaded file across reruns.
4487 ``st.file_uploader`` keeps the same ``UploadedFile`` (and ``file_id``) for a
4488 given upload until it's replaced, so keying on it lets us parse the file
4489 *once* instead of on every rerun."""
4490 return (
4491 getattr(uploaded, "file_id", None),
4492 getattr(uploaded, "name", None),
4493 getattr(uploaded, "size", None),
4494 )
4497@st.cache_data(show_spinner="Reading uploaded data…", show_time=True)
4498def _read_uploaded_table_cached(
4499 _uploaded, file_key, kind=None, chosen=(), text_column=None, identity=()
4500) -> pd.DataFrame:
4501 try:
4502 _uploaded.seek(0)
4503 except Exception:
4504 pass
4505 if kind is None:
4506 return stamp_source(read_table(_uploaded))
4507 # PERF-6: parse only the columns the mapping, the registry and the user's
4508 # own picks need. `kind` and `chosen` are part of the cache key, so naming
4509 # a new column simply re-reads the file under the new plan. `kind` also
4510 # picks a mixed zip's members (#374 F3).
4511 header = read_table_columns(_uploaded, kind=kind)
4512 plan = upload_read_plan(
4513 header, kind, chosen=chosen, text_column=text_column, identity=identity
4514 )
4515 return stamp_source(read_table(_uploaded, plan=plan, kind=kind))
4518@st.cache_data(show_spinner="Reading uploaded data…", show_time=True)
4519def _read_uploaded_tables_cached(
4520 _uploaded_list, file_keys, kind=None, chosen=(), text_column=None, identity=()
4521) -> pd.DataFrame:
4522 for f in _uploaded_list:
4523 try:
4524 f.seek(0)
4525 except Exception:
4526 pass
4527 plan_for = None
4528 if kind is not None:
4530 def plan_for(header):
4531 return upload_read_plan(
4532 header, kind, chosen=chosen, text_column=text_column, identity=identity
4533 )
4535 return stamp_source(read_tables(list(_uploaded_list), plan_for=plan_for, kind=kind))
4538#: Session keys naming a source column the user has picked: every mapping
4539#: dropdown (``col_map_<table>_<field>``) and the wizard's per-table
4540#: extra-keeps pickers (``wizard_keep_<prefix>`` — UX-114; was one cross-table
4541#: ``wizard_keep_extra`` key before). A composite trial id stores a *list*, so
4542#: both shapes are read.
4543_CHOSEN_COLUMN_KEYS = ("col_map_", "wizard_keep_")
4546def _columns_chosen_in_state(state, header) -> set:
4547 """Source columns of ``header`` the user has already named (PERF-6).
4549 Swept out of session state rather than read field by field: the mapping
4550 keys are per-table *and* per-field, and a composite trial id stores a list,
4551 so matching names against the header is both simpler and robust to a key
4552 this function has never heard of. Names belonging to the *other* upload box
4553 — or left over from a previous dataset — aren't columns of this table, so
4554 the header filter drops them.
4555 """
4556 columns = set(header)
4557 chosen: set = set()
4558 for key, value in state.items():
4559 if not str(key).startswith(_CHOSEN_COLUMN_KEYS):
4560 continue
4561 values = value if isinstance(value, (list, tuple, set)) else [value]
4562 chosen.update(v for v in values if isinstance(v, str) and v in columns)
4563 return chosen
4566def upload_read_plan(
4567 header, kind: str, *, chosen=(), text_column: str | None = None, identity=()
4568) -> ReadPlan:
4569 """Plan an uploaded table's read from its header (PERF-6, decision 2a).
4571 The mapping is auto-proposed from the column names, so the plan exists
4572 before the user has touched anything; ``chosen`` folds back in the columns
4573 they *have* named, which is what keeps a hand-picked mapping or a kept extra
4574 from being dropped. A column named later simply changes the plan, and the
4575 read runs again against the new one. ``text_column`` is the user's own
4576 word-text pick, read verbatim in place of the proposed one (BUG-53), and
4577 ``identity`` their own id-column picks, read as text (BUG-59).
4578 """
4579 propose = propose_word_schema if kind == "words" else propose_fix_schema
4580 registry = WORD_OPTIONAL_FIELDS if kind == "words" else FIX_OPTIONAL_FIELDS
4581 names = list(header)
4582 return plan_table_read(
4583 names,
4584 propose(pd.DataFrame(columns=names)),
4585 registry,
4586 keep_columns=set(chosen),
4587 text_column=text_column,
4588 identity_columns=identity,
4589 )
4592def _upload_header(uploaded, *, multi: bool, kind: str | None = None) -> list:
4593 """Every column name across an upload, in first-seen order (PERF-6).
4595 The *union*, not the first file's: one upload is commonly one file per
4596 participant, and an export can gain or lose a column between them
4597 (``read_tables``: "fields absent from a file become NaN"). Resolving the
4598 user's chosen columns against only the first header would silently drop a
4599 column that lives in a later file, and the mapping dropdowns would not
4600 offer it at all.
4601 """
4602 sources = list(uploaded) if multi else [uploaded]
4603 header: list = []
4604 for source in sources:
4605 columns = _upload_columns_cached(source, _uploaded_file_key(source), kind)
4606 header.extend(c for c in columns if c not in header)
4607 return header
4610@st.cache_data(show_spinner=False, max_entries=64)
4611def _upload_columns_cached(_uploaded, file_key, kind: str | None = None) -> list:
4612 """One uploaded file's column names, read once per file (PERF-6's header pass).
4614 Keyed like the planned read. A delimited file's header is cheap, but a
4615 workbook or a zipped Parquet / Feather / Excel member has no header-only
4616 read — :func:`data.read_table_columns` parses it whole — so an uncached pass
4617 re-parsed the file on every rerun of the wizard: 1.4 s a click on a full
4618 ``.xls`` sheet, and a second decompressed copy of a large zip held at once.
4619 """
4620 return read_table_columns(_uploaded, kind=kind)
4623@st.cache_data(show_spinner=False, max_entries=64)
4624def _zip_split_cached(_uploaded, file_key, kind: str | None) -> str:
4625 """The note an upload row shows when its zip mixes fixation and
4626 interest-area reports (#374 F3): which members it used, which it left out.
4627 Empty for anything else."""
4628 try:
4629 return zip_member_split(_uploaded, kind).message()
4630 except Exception: # the read itself reports an unreadable archive
4631 return ""
4634@st.cache_data(show_spinner=False, max_entries=32)
4635def _upload_sample_cached(_uploaded, file_key, kind: str | None) -> pd.DataFrame:
4636 """The first rows of one upload, every column parsed (#374 F13)."""
4637 return read_table_sample(_uploaded, kind=kind)
4640def upload_sample(state_prefix: str, kind: str | None) -> pd.DataFrame:
4641 """A sample of the first file uploaded under ``state_prefix`` — what the
4642 wizard judges a column the planned read left out by. Empty without one."""
4643 uploaded = st.session_state.get(f"{state_prefix}_upload")
4644 files = uploaded if isinstance(uploaded, (list, tuple)) else [uploaded]
4645 first = next((f for f in files if f is not None), None)
4646 if first is None or not hasattr(first, "read"):
4647 return pd.DataFrame()
4648 return _upload_sample_cached(first, _uploaded_file_key(first), kind)
4651def upload_zip_notes(uploaded, kind: str | None) -> list[str]:
4652 """One note per zip in an upload that left members out for ``kind``."""
4653 if kind is None or not uploaded:
4654 return []
4655 files = uploaded if isinstance(uploaded, (list, tuple)) else [uploaded]
4656 notes = []
4657 for f in files:
4658 if str(getattr(f, "name", "")).lower().endswith(".zip"):
4659 note = _zip_split_cached(f, _uploaded_file_key(f), kind)
4660 if note:
4661 notes.append(note)
4662 return notes
4665def _uploaded_header(state_prefix: str) -> list:
4666 """The full column list of the table uploaded under ``state_prefix``.
4668 PERF-6 narrows the *rows* an upload parses, never the column names: the
4669 mapping dropdowns and the wizard's "Additional fields to keep" picker still
4670 offer every column in the file, and naming one adds it to the plan. Empty
4671 when nothing is uploaded, or on a path that reads the table whole.
4672 """
4673 return list(st.session_state.get(f"{state_prefix}_header") or [])
4676def _read_uploaded_frame(
4677 *,
4678 uploader_label: str,
4679 upload_help: str,
4680 state_prefix: str,
4681 multi: bool,
4682 container=None,
4683 kind: str | None = None,
4684 label_visibility: str = "visible",
4685) -> pd.DataFrame:
4686 """Render one upload box and return its (concatenated) frame.
4688 Renders into ``container`` — the setup wizard's own step, or the 🗂️ Data
4689 page's upload slot. Empty frame when nothing is
4690 uploaded. The file parse is cached on the upload's identity (see
4691 ``_uploaded_file_key``) so a large uploaded table is read once, not re-parsed
4692 on every rerun. Isolated from the mapping render so tests can inject frames
4693 without a real upload (AppTest can't drive ``st.file_uploader``).
4695 ``label_visibility="collapsed"`` (UX-113) lets a caller draw its own title
4696 above the box — e.g. via ``controls.inline_field_label``'s dotted-underline
4697 hover format, matching the mapping steps' field titles — instead of
4698 Streamlit's own label + native (~1s) help tooltip. The widget still gets the
4699 real ``uploader_label``/``upload_help`` as its accessible name and help; only
4700 where they are drawn changes.
4701 """
4702 host = container if container is not None else st.container()
4703 uploaded = host.file_uploader(
4704 uploader_label,
4705 type=_UPLOAD_TYPES,
4706 accept_multiple_files=multi,
4707 key=f"{state_prefix}_upload",
4708 help=upload_help,
4709 label_visibility=label_visibility,
4710 max_upload_size=upload_limit_mb(),
4711 )
4712 if not uploaded:
4713 return pd.DataFrame()
4714 # BUG-5: a large upload parses/normalizes into several in-memory copies that
4715 # can OOM-kill the ~1 GB hosted demo (no traceback). Warn and require an
4716 # explicit opt-in before parsing.
4717 #
4718 # DATA-22 review: only on the *hosted* demo. Running locally there is no such
4719 # ceiling — the warning was pure noise, and the "Load it anyway" tick was a
4720 # step between the user and their own data on their own machine. Same
4721 # loopback test the wizard's "run locally" tip uses.
4722 if upload_exceeds_limit(uploaded) and not is_loopback_url(
4723 str(getattr(st.context, "url", "") or "")
4724 ):
4725 mb = uploaded_files_total_bytes(uploaded) / (1024 * 1024)
4726 host.warning(
4727 f"This upload is **{mb:.0f} MB**. On the hosted demo (~1 GB RAM), "
4728 "parsing a corpus this large can exhaust memory and crash the app. "
4729 f"For big corpora, use the [desktop app]({CITATION['desktop_url']}) "
4730 "or `pip install scanpath-studio`, or upload a subset (e.g. a few "
4731 "participants)."
4732 )
4733 if not host.checkbox(
4734 "Load it anyway",
4735 key=f"{state_prefix}_load_large",
4736 help="Parse this large upload regardless. Safe on a local machine "
4737 "with enough RAM; may crash the memory-limited hosted demo.",
4738 ):
4739 return pd.DataFrame()
4740 # PERF-6: the header pass is cheap and its answer is what both the plan and
4741 # the wizard's column pickers are built from, so it happens first and is
4742 # stashed for `_uploaded_header`. `chosen` is sorted into a tuple because it
4743 # rides in the cache key.
4744 # BUG-55: a file the readers refuse — an empty file, a corrupt archive or
4745 # workbook — is the user's to fix, so it is said in the box
4746 # that took it, the way the metadata uploaders already do, instead of a
4747 # traceback over the whole page.
4748 try:
4749 frame = _read_upload(uploaded, state_prefix, multi=multi, kind=kind)
4750 except Exception as exc: # unreadable file — say so, keep the page
4751 logging.getLogger(__name__).warning(
4752 "Could not read upload %s", state_prefix, exc_info=True
4753 )
4754 st.session_state.pop(f"{state_prefix}_header", None)
4755 files = uploaded if multi else [uploaded]
4756 names = ", ".join(str(getattr(f, "name", "the file")) for f in files)
4757 host.error(
4758 f"Couldn't read **{names}**: {exc}. Check it is a table file with one "
4759 "header row."
4760 )
4761 return pd.DataFrame()
4762 # BUG-103: this upload's own ID, before the wizard derives anything from it.
4763 adopt_source(frame)
4764 return frame
4767def _read_upload(uploaded, state_prefix: str, *, multi: bool, kind) -> pd.DataFrame:
4768 """The header pass and the (cached) planned read behind one upload box."""
4769 header: list = []
4770 chosen: tuple = ()
4771 text_column = None
4772 identity: tuple = ()
4773 if kind is not None:
4774 header = _upload_header(uploaded, multi=multi, kind=kind)
4775 chosen = tuple(sorted(_columns_chosen_in_state(st.session_state, header)))
4776 # BUG-53: the word-text column the user mapped by hand (the mapping
4777 # widget's own key) is the one to read verbatim, not the proposed one.
4778 picked = st.session_state.get(f"{state_prefix}_text")
4779 if kind == "words" and isinstance(picked, str) and picked in header:
4780 text_column = picked
4781 # BUG-59: likewise the id columns picked by hand, read as text so a
4782 # zero-padded id keeps its zeros.
4783 identity = _picked_columns(state_prefix, IDENTITY_SCHEMA_FIELDS, header)
4784 st.session_state[f"{state_prefix}_header"] = header
4785 if multi:
4786 return _read_uploaded_tables_cached(
4787 uploaded,
4788 tuple(_uploaded_file_key(f) for f in uploaded),
4789 kind=kind,
4790 chosen=chosen,
4791 text_column=text_column,
4792 identity=identity,
4793 )
4794 return _read_uploaded_table_cached(
4795 uploaded,
4796 _uploaded_file_key(uploaded),
4797 kind=kind,
4798 chosen=chosen,
4799 text_column=text_column,
4800 identity=identity,
4801 )
4804def _picked_columns(state_prefix: str, fields, header) -> tuple:
4805 """The header columns the mapping widgets for ``fields`` currently name."""
4806 columns = set(header)
4807 picked: list = []
4808 for name in fields:
4809 value = st.session_state.get(f"{state_prefix}_{name}")
4810 values = value if isinstance(value, (list, tuple)) else [value]
4811 picked += [v for v in values if isinstance(v, str) and v in columns]
4812 return tuple(sorted(set(picked)))
4815def load_raw_gaze_data(data_choice: str, *, host=None, notices=None) -> pd.DataFrame:
4816 """Load and normalize optional raw gaze data (millisecond-level eye positions).
4818 Raw gaze data provides finer temporal resolution than fixation-level data
4819 and enables overlay visualizations showing continuous gaze paths.
4821 Args:
4822 data_choice: The selected data source (e.g. ``DEMO_CHOICE`` loads the
4823 bundled sample gaze; other built-in sources have none). The Upload
4824 source and stored datasets carry their own raw gaze, so ``main``
4825 doesn't call this for them.
4826 host: Where the optional uploader + its column mapping render — the
4827 *Data location* section of the 🗂️ Data page (DATA-26).
4828 notices: Where the "raw gaze ignored" warnings render. Deliberately the
4829 strip under the menu bar, not ``host``: a warning on a page the user
4830 isn't looking at is invisible, and that strip is on every page.
4832 Returns:
4833 Normalized raw gaze DataFrame with canonical columns, or empty DataFrame
4834 if not available or schema inference fails
4836 Canonical Columns (raw gaze):
4837 participant_id, trial_id, x, y, timestamp_ms (optional: text)
4839 UI Effects:
4840 - Renders optional file uploader for "Upload csv tables" mode
4841 - Shows warning if schema inference fails
4842 - Shows info message if sample data unavailable
4843 """
4844 raw_gaze_df = pd.DataFrame()
4845 cfg = host if host is not None else st.container()
4846 warn = notices if notices is not None else st.container()
4848 if data_choice in (SYNTHETIC_CHOICE, PUBLIC_DATASETS_CHOICE):
4849 # Neither the synthetic trial nor the public corpora ship raw gaze;
4850 # skip the uploader entirely.
4851 return raw_gaze_df
4853 # PERF-11: raw gaze is recorded at up to 1000 Hz, so a real table is
4854 # millions of rows — and both branches below re-read and re-normalized it on
4855 # every rerun (~1.4 s per click at 1M rows). `frame_cache` keeps the result
4856 # while its inputs hold and hands back the same object, as for the corpus.
4857 if data_choice == DEMO_CHOICE:
4859 def _demo_raw_gaze() -> tuple[pd.DataFrame, dict | None, bool]:
4860 sample = load_sample_raw_gaze()
4861 if sample.empty:
4862 return sample, None, False
4863 schema = infer_raw_gaze_schema(sample)
4864 if not schema:
4865 return pd.DataFrame(), None, True
4866 return normalize_raw_gaze(sample, schema), schema, False
4868 raw_gaze_df, raw_gaze_schema, unmappable = frame_cache(
4869 "raw_gaze", ("demo",), _demo_raw_gaze
4870 )
4871 if raw_gaze_schema:
4872 # The raw sample is read inside the cached builder; its loader is
4873 # cached too, so asking it again costs a copy of a 2k-row sample.
4874 _stash_active_mapping(
4875 "raw_gaze", raw_gaze_schema, load_sample_raw_gaze().columns
4876 )
4877 elif unmappable:
4878 warn.warning("The bundled demo's raw-gaze samples couldn't be read.")
4879 else:
4880 uploaded_raw_gaze = cfg.file_uploader(
4881 "Raw gaze table (optional)",
4882 type=["csv", "parquet", "feather", "zip"],
4883 help=(
4884 "Optional: one row per gaze sample — participant, trial, x, y "
4885 "and, if recorded, a timestamp."
4886 ),
4887 max_upload_size=upload_limit_mb(),
4888 )
4889 if uploaded_raw_gaze:
4890 upload_key = (uploaded_raw_gaze.file_id, uploaded_raw_gaze.size)
4891 try:
4892 raw_gaze_df = frame_cache(
4893 "raw_gaze_upload",
4894 upload_key,
4895 lambda: read_table(uploaded_raw_gaze),
4896 )
4897 except Exception as exc: # unreadable file — say so, keep the page
4898 cfg.error(f"Couldn't read **{uploaded_raw_gaze.name}**: {exc}")
4899 return pd.DataFrame()
4900 proposed = propose_raw_gaze_schema(raw_gaze_df)
4901 initial_problems = validate_raw_gaze_schema(proposed)
4902 with cfg:
4903 raw_gaze_schema = column_mapping_ui(
4904 raw_gaze_df,
4905 table_label="Raw gaze",
4906 state_key_prefix="col_map_raw_gaze",
4907 field_specs=RAW_GAZE_FIELD_SPECS,
4908 proposed=proposed,
4909 problems=initial_problems,
4910 # BUG-32: the same source key `main` scopes the tables by.
4911 dataset=(
4912 data_choice,
4913 st.session_state.get("public_dataset_choice"),
4914 ),
4915 )
4916 problems = validate_raw_gaze_schema(raw_gaze_schema)
4917 if problems:
4918 warn.warning("Raw gaze ignored — " + "; ".join(problems))
4919 raw_gaze_df = pd.DataFrame()
4920 else:
4921 _stash_active_mapping("raw_gaze", raw_gaze_schema, raw_gaze_df.columns)
4922 source = raw_gaze_df
4923 raw_gaze_df = frame_cache(
4924 "raw_gaze",
4925 (upload_key, _schema_key(raw_gaze_schema)),
4926 lambda: normalize_raw_gaze(source, raw_gaze_schema),
4927 )
4929 return raw_gaze_df
4932# -----------------------------------------------------------------------------
4933# Data-source resolution + the panels the top menu bar hosts
4934#
4935# `_sidebar_group` is gone with the sidebar (UX-38): each former group is its own
4936# popover on the menu bar (see `menu.render_top_menu`), and the popover's trigger
4937# label is the group heading. Nothing left to title.
4938# -----------------------------------------------------------------------------
4941#: Every built-in token :func:`resolve_data_source` can put in the picker,
4942#: whatever this run's gates — the ones a user's dataset may never be named
4943#: (:func:`reserved_source_names`). A name that shadows one gives the picker a
4944#: duplicate option, hijacks the built-in's load branch, and (DATA-47/48) shares
4945#: its metadata tables and annotations.
4946BUILTIN_SOURCE_CHOICES = (
4947 ONESTOP_CHOICE,
4948 MULTIPLEYE_BUNDLE_CHOICE,
4949 DEMO_CHOICE,
4950 MANUAL_SAMPLE_CHOICE,
4951 SYNTHETIC_CHOICE,
4952 AUTHOR_CHOICE,
4953 UPLOAD_CHOICE,
4954 PUBLIC_DATASETS_CHOICE,
4955)
4958def reserved_source_names() -> frozenset[str]:
4959 """Every built-in data-source label: the fixed tokens and every corpus."""
4960 return (
4961 frozenset(BUILTIN_SOURCE_CHOICES)
4962 | frozenset(PUBLIC_DATASET_REGISTRY)
4963 | frozenset(public_dataset_registry())
4964 )
4967def resolve_data_source(host=None) -> str:
4968 """Resolve the active data source (renders no picker widget — UX-25).
4970 Returns the selected source: ``DEMO_CHOICE`` ("Bundled Demo"), a stored
4971 uploaded dataset's name, ``ONESTOP_CHOICE`` / ``PUBLIC_DATASETS_CHOICE`` when
4972 available, ``SYNTHETIC_CHOICE``, or ``UPLOAD_CHOICE``
4973 while the "➕ Add data" wizard is active. Switching to a stored dataset reloads
4974 it from session (no re-upload). Manual authoring is opened by the + menu;
4975 its draft joins the list once opened, including through an old deep link.
4977 **UX-25** moved the *visible* picker out of the sidebar and onto the main
4978 view's "Filter by" row (:func:`render_data_source_picker`). The picker has to
4979 render inside the tab, i.e. long after the data is loaded, so this function
4980 keeps its position at the top of ``main`` and stays the resolver: it applies
4981 the pre-widget ``_pending_source_choice`` seam, heals a stale selection, and
4982 publishes the entry list the picker renders from (``_data_source_entries``).
4983 ``data_source_choice`` remains the canonical key (``?source=…`` deep links and
4984 the wizard's finalize / cancel path both write it).
4986 ``host`` is the Data page's *Data source* slot (DATA-26). The one thing this
4987 function *does* render — the wizard's "✕ Cancel" bar, which stands in for the
4988 picker while an upload is being added — goes there, so it takes the picker's
4989 place on the page rather than appearing above it in the bare main area.
4990 """
4991 # Apply a programmatic source switch (the wizard's finalize / Cancel, or the
4992 # main-view picker's on_change) BEFORE anything reads data_source_choice. It
4993 # rides a plain key, not a widget value, so the browser never reconciles it
4994 # away — assigning data_source_choice inline and rerunning is unreliable
4995 # because the widget's frontend value can overwrite it on the rerun (works in
4996 # AppTest, not in a real browser). Callbacks run before the script body, so a
4997 # pick made in the tab still takes effect on the very next run.
4998 pending = st.session_state.pop("_pending_source_choice", None)
4999 if pending is not None:
5000 st.session_state["data_source_choice"] = pending
5001 # A real source was chosen (finalize / cancel) → leave the wizard.
5002 st.session_state["_show_upload_wizard"] = False
5004 # The upload wizard is tracked by a plain flag, not by parking UPLOAD_CHOICE
5005 # in the radio key (which Streamlit would garbage-collect mid-wizard — see
5006 # _enter_add_data_wizard). The legacy ``data_source_choice == UPLOAD_CHOICE``
5007 # is still honoured so AppTests / `?source=upload` deep links can open the
5008 # wizard directly. While it's open the wizard owns the page (the "Filter by"
5009 # row never renders), so the way out is rendered here, at the top of it.
5010 if (
5011 st.session_state.get("_show_upload_wizard")
5012 or st.session_state.get("data_source_choice") == UPLOAD_CHOICE
5013 ):
5014 # UX-66: the caption is gone (the sticky bar's title says where you are)
5015 # and ✕ Cancel rides that bar — `wizard._render_data_setup` reserves the
5016 # slot, and the wizard renders *after* this, so on the very first run of
5017 # a fresh wizard the slot does not exist yet and it falls back to here.
5018 # UX-66: ✕ Cancel moved onto the wizard's sticky bar, which is the one
5019 # row that stays on screen — the way out used to scroll away with the
5020 # page. It is rendered by `wizard._render_data_setup` via
5021 # `leave_add_data_wizard` below; nothing is drawn here, because this
5022 # function runs *before* the wizard and a container reserved now would
5023 # belong to the previous run.
5024 return UPLOAD_CHOICE
5026 # DATA-9: one **flat** source picker. Every source is a single entry tagged by
5027 # kind — 🧪 demo · 🔒 private (your uploads + local env bundles) · 🌐 public —
5028 # instead of a "Public datasets" category that then needed a second selectbox.
5029 # `data_source_choice` stays the canonical key, but for a public corpus the
5030 # entry's token IS the registry label; the return value resolves it back to
5031 # PUBLIC_DATASETS_CHOICE (+ public_dataset_choice) so the load path is unchanged.
5032 uploaded = list(st.session_state.get("_datasets", {}).keys())
5033 entries: list[str] = []
5034 kinds: dict[str, str] = {}
5035 if onestop_data_dir() is not None:
5036 entries.append(ONESTOP_CHOICE)
5037 kinds[ONESTOP_CHOICE] = "🔒"
5038 if multipleye_bundle_dir() is not None:
5039 entries.append(MULTIPLEYE_BUNDLE_CHOICE)
5040 kinds[MULTIPLEYE_BUNDLE_CHOICE] = "🔒"
5041 entries.append(DEMO_CHOICE)
5042 kinds[DEMO_CHOICE] = "🧪"
5043 entries.append(MANUAL_SAMPLE_CHOICE)
5044 kinds[MANUAL_SAMPLE_CHOICE] = "✏️"
5045 if (
5046 debug_enabled()
5047 or st.session_state.get("data_source_choice") == SYNTHETIC_CHOICE
5048 ):
5049 entries.append(SYNTHETIC_CHOICE)
5050 kinds[SYNTHETIC_CHOICE] = "🧪"
5051 if st.session_state.get(
5052 "data_source_choice"
5053 ) == AUTHOR_CHOICE or AUTHOR_CHOICE in st.session_state.get(
5054 "_manual_scanpath_drafts", {}
5055 ):
5056 entries.append(AUTHOR_CHOICE)
5057 kinds[AUTHOR_CHOICE] = "✏️"
5058 for name in uploaded:
5059 entries.append(name)
5060 kinds[name] = (
5061 "✏️" if st.session_state["_datasets"][name].get("authoring") else "🔒"
5062 )
5063 # DATA-27 (Task 11R): every prepared benchmark corpus is in here as its own
5064 # 🌐 entry, exactly like the built-ins — `public_dataset_registry()` composes
5065 # the two. Resolved once and reused below so the whole run agrees on one
5066 # snapshot of a registry that depends on a directory the user can change.
5067 registry = public_dataset_registry() if public_datasets_enabled() else {}
5068 for label in registry:
5069 entries.append(label)
5070 kinds[label] = "🌐"
5071 # Removing an app-owned/public source means removing it from this session's
5072 # available list, not deleting packaged files or a public corpus. Keep the
5073 # stable token intact for links and loader dispatch; the ordinary stale-
5074 # selection healing below moves away from a source that was just hidden.
5075 hidden = set(st.session_state.get(HIDDEN_DATASETS_KEY) or [])
5076 entries = [token for token in entries if token not in hidden]
5077 kinds = {token: kind for token, kind in kinds.items() if token in entries}
5078 if not entries:
5079 # Never strand the app without a loadable source. This can only happen
5080 # after the user has removed every row one by one in the same session.
5081 hidden.discard(DEMO_CHOICE)
5082 st.session_state[HIDDEN_DATASETS_KEY] = sorted(hidden)
5083 entries = [DEMO_CHOICE]
5084 kinds = {DEMO_CHOICE: "🧪"}
5086 # Migrate a legacy `PUBLIC_DATASETS_CHOICE` selection (old saved state / deep
5087 # link / the former category radio) to the concrete corpus token so it lands on
5088 # the right entry. Falls back to the first public corpus (not the demo) when no
5089 # corpus was remembered, preserving the old "Public datasets → first corpus".
5090 if st.session_state.get("data_source_choice") == PUBLIC_DATASETS_CHOICE:
5091 corpus = st.session_state.get("public_dataset_choice")
5092 if corpus not in registry:
5093 corpus = next(iter(registry), None)
5094 st.session_state["data_source_choice"] = corpus or entries[0]
5096 # Heal a stale/invalid selection (e.g. a removed dataset) so the picker never
5097 # errors on an option that is no longer in the list. A stale benchmark
5098 # *corpus* label (the bundle directory was repointed, or that corpus was
5099 # removed from it) lands on another added corpus in preference to
5100 # `entries[0]` (the demo) whenever one is reachable (N2).
5101 stale = str(st.session_state.get("data_source_choice") or "")
5102 if stale not in entries:
5103 healed = ""
5104 if stale.endswith(BENCHMARK_LABEL_SUFFIX):
5105 healed = next(
5106 (
5107 label
5108 for label, spec in registry.items()
5109 if spec.get("benchmark_dataset")
5110 ),
5111 "",
5112 )
5113 st.session_state["data_source_choice"] = healed or entries[0]
5114 choice = st.session_state["data_source_choice"]
5116 # Publish what the main-view picker renders from. It runs inside the tab,
5117 # after this; recomputing the list there would duplicate the registry /
5118 # stored-dataset logic above (and could disagree with the healed selection).
5119 st.session_state["_data_source_entries"] = entries
5120 st.session_state["_data_source_kinds"] = kinds
5121 st.session_state["_data_source_uploaded"] = uploaded
5123 # Resolve a public-corpus token back to the canonical PUBLIC_DATASETS_CHOICE so
5124 # every downstream consumer (load dispatch, monitor, filter/col-map reset keys)
5125 # is unchanged; the chosen corpus rides public_dataset_choice as before.
5126 if choice in registry:
5127 st.session_state["public_dataset_choice"] = choice
5128 return PUBLIC_DATASETS_CHOICE
5129 return choice
5132def _on_data_source_pick() -> None:
5133 """Route the main-view picker's choice through the pre-widget seam (UX-25).
5135 An ``on_change`` callback: it runs before the rerun's script body, so
5136 ``resolve_data_source`` — which pops ``_pending_source_choice`` at the
5137 top of ``main`` — applies the new source on the *same* run that renders it.
5138 The picker rides its own widget key (``data_source_picker``) rather than
5139 writing ``data_source_choice`` directly, so a deep link / saved config can
5140 keep assigning the canonical key without the widget reconciling it away.
5141 """
5142 picked = st.session_state.get("data_source_picker")
5143 if picked:
5144 if picked == AUTHOR_CHOICE:
5145 _remember_authoring_return()
5146 st.session_state["_pending_source_choice"] = picked
5149def _remember_authoring_return() -> None:
5150 source = st.session_state.get("data_source_choice", DEMO_CHOICE)
5151 if source not in (AUTHOR_CHOICE, MANUAL_SAMPLE_CHOICE):
5152 st.session_state["_author_return_source"] = source
5155#: The authoring source whose editor is open. ``AUTHOR_CHOICE`` is always an
5156#: editor; the synthetic sample is a dataset that is *shown* like any other until
5157#: its ✏️ Edit button arms this key. It is dropped as soon as another source is
5158#: active, so coming back to the sample shows it rather than reopening the editor.
5159_AUTHOR_EDITING_KEY = "_author_editing"
5161#: The synthetic sample's stimulus until it has been edited.
5162_MANUAL_SAMPLE_TEXT = "The cat sat\non the mat."
5165def _authoring_editor_open(data_choice: str) -> bool:
5166 """Whether ``data_choice`` renders the authoring editor instead of a view."""
5167 return data_choice == AUTHOR_CHOICE or (
5168 data_choice == MANUAL_SAMPLE_CHOICE
5169 and st.session_state.get(_AUTHOR_EDITING_KEY) == MANUAL_SAMPLE_CHOICE
5170 )
5173def _edit_manual_sample() -> None:
5174 """Open the authoring editor on the synthetic sample (its ✏️ Edit button)."""
5175 st.session_state[_AUTHOR_EDITING_KEY] = MANUAL_SAMPLE_CHOICE
5176 st.session_state["main_nav"] = _VIEW_SCANPATH
5179def _manual_sample_document() -> tuple[pd.DataFrame, pd.DataFrame, dict]:
5180 """The synthetic sample's ``(words, events, layout)``, drawn from its draft.
5182 The seed text until it has been edited, the draft afterwards — the same
5183 document the editor shows, so viewing the sample never needs the editor.
5184 """
5185 from scanpath_studio.authoring import DEFAULT_LAYOUT, default_events, layout_text
5187 draft = st.session_state.get("_manual_scanpath_drafts", {}).get(
5188 MANUAL_SAMPLE_CHOICE
5189 )
5190 if draft is None:
5191 layout = dict(DEFAULT_LAYOUT)
5192 words = layout_text(_MANUAL_SAMPLE_TEXT, **layout)
5193 return words, default_events(words), layout
5194 text, layout, events = draft
5195 layout = {**DEFAULT_LAYOUT, **layout}
5196 return layout_text(text, **layout), events, layout
5199def _manual_sample_frames() -> tuple[pd.DataFrame, pd.DataFrame]:
5200 """The synthetic sample's words and fixations (see `_manual_sample_document`)."""
5201 from scanpath_studio.authoring import authored_fixations
5203 words, events, _layout = _manual_sample_document()
5204 return words, authored_fixations(words, events)
5207def _manual_sample_canvas() -> tuple[int, int]:
5208 """The canvas the authoring editor draws the sample on, as a figure size.
5210 The sample declares its own screen: without it the figure inherits the
5211 previous source's (the demo's 2560 × 1440) and six words sit in a corner.
5212 """
5213 words, _events, layout = _manual_sample_document()
5214 height = 480
5215 if not words.empty:
5216 height = max(
5217 height,
5218 int(words["y"].max() + words["height"].max() + layout["margin"]),
5219 )
5220 return int(layout["canvas_width"]), height
5223def _cancel_authoring() -> None:
5224 if st.session_state.get(_AUTHOR_EDITING_KEY) == MANUAL_SAMPLE_CHOICE:
5225 # Back out of the sample's editor to the sample itself.
5226 st.session_state.pop(_AUTHOR_EDITING_KEY, None)
5227 st.session_state["main_nav"] = _VIEW_SCANPATH
5228 return
5229 st.session_state["_pending_source_choice"] = st.session_state.get(
5230 "_author_return_source", DEMO_CHOICE
5231 )
5232 st.session_state["main_nav"] = _VIEW_SCANPATH
5235#: ``{source: {fixation_id: (word_id, word)}}`` — target words a stimulus edit
5236#: left out of date (`authoring.stale_target_words`). Flagged on the editor,
5237#: never rewritten.
5238_AUTHOR_STALE_TARGETS_KEY = "_author_stale_targets"
5239#: ``{source: (text, layout, events)}`` — the draft before the last change that
5240#: removed, moved or retimed fixations (`authoring.destructive_change`). One
5241#: step, swapped with the current draft by **Restore previous draft**.
5242_AUTHOR_PREVIOUS_DRAFT_KEY = "_author_previous_drafts"
5243#: What the downloaded authoring file is called — the name Share → Code's
5244#: snippet reads it by (`url_state._snippet_source`).
5245AUTHORING_FILE_NAME = "scanpath.json"
5248def _load_author_draft(source: str, draft: tuple) -> None:
5249 """Put ``draft`` — ``(text, layout, events)`` — on the authoring screen.
5251 Written before the widgets render (a callback), and the table remounts from
5252 the new events rather than replaying its old edits over them (BUG-19)."""
5253 text, layout, events = draft
5254 st.session_state["author_text"] = text
5255 st.session_state["_author_layout"] = dict(layout)
5256 st.session_state["_authored_events_frame"] = events.copy()
5257 st.session_state["_author_text_for_events"] = text
5258 st.session_state["_author_selected_fixation"] = None
5259 st.session_state["_author_events_editor_revision"] = (
5260 int(st.session_state.get("_author_events_editor_revision", 0)) + 1
5261 )
5262 st.session_state.setdefault(_AUTHOR_STALE_TARGETS_KEY, {}).pop(source, None)
5265def _restore_previous_author_draft(source: str) -> None:
5266 """Swap the current draft with the one before the last destructive edit.
5268 Pressing it again swaps back, so a restore is never itself a loss."""
5269 previous = st.session_state.get(_AUTHOR_PREVIOUS_DRAFT_KEY, {}).get(source)
5270 drafts = st.session_state.setdefault("_manual_scanpath_drafts", {})
5271 if previous is None:
5272 return
5273 current = drafts.get(source)
5274 _load_author_draft(source, previous)
5275 drafts[source] = previous
5276 if current is not None:
5277 st.session_state[_AUTHOR_PREVIOUS_DRAFT_KEY][source] = current
5280def _reset_author_fixations(source: str, words: pd.DataFrame) -> None:
5281 """Replace the fixations with one per word — the explicit regeneration.
5283 The draft it replaces becomes the previous draft at the end of the run
5284 (a destructive change), so **Restore previous draft** brings it back."""
5285 from scanpath_studio.authoring import default_events
5287 st.session_state["_authored_events_frame"] = default_events(words)
5288 st.session_state["_author_selected_fixation"] = None
5289 st.session_state["_author_events_editor_revision"] = (
5290 int(st.session_state.get("_author_events_editor_revision", 0)) + 1
5291 )
5292 st.session_state.setdefault(_AUTHOR_STALE_TARGETS_KEY, {}).pop(source, None)
5295def _save_authored_dataset(name_key: str) -> None:
5296 from scanpath_studio.wizard import _safe_dataset_name
5298 payload = st.session_state.pop("_author_save_payload", None)
5299 if payload is None:
5300 return
5301 requested = str(st.session_state.get(name_key) or "My scanpath").strip()
5302 if requested in (AUTHOR_CHOICE, MANUAL_SAMPLE_CHOICE):
5303 requested += " (authored)"
5304 name = _safe_dataset_name(requested)
5305 st.session_state.setdefault("_datasets", {})[name] = payload
5306 # DATA-48: the draft's annotations were made on the scanpath being saved,
5307 # so they become the saved dataset's. The name is safe, so it holds none.
5308 if st.session_state.get("data_source_choice") == AUTHOR_CHOICE:
5309 import scanpath_studio.annotations as _annotations
5311 _annotations.rename_dataset(st.session_state, AUTHOR_CHOICE, name)
5312 st.session_state["_pending_source_choice"] = name
5313 st.session_state["main_nav"] = _VIEW_SCANPATH
5314 st.session_state["setup_complete"] = True
5317def _enter_manual_dataset() -> None:
5318 """Open (or resume) the manual editor through the ordinary source switch."""
5319 _remember_authoring_return()
5320 hidden = list(st.session_state.get(HIDDEN_DATASETS_KEY) or [])
5321 if AUTHOR_CHOICE in hidden:
5322 hidden.remove(AUTHOR_CHOICE)
5323 st.session_state[HIDDEN_DATASETS_KEY] = hidden
5324 st.session_state["_pending_source_choice"] = AUTHOR_CHOICE
5325 st.session_state["main_nav"] = _VIEW_SCANPATH
5328def leave_add_data_wizard() -> None:
5329 """Abandon the add-dataset wizard and go back to the previous source.
5331 Split out of the picker for UX-66, which moved ✕ Cancel onto the wizard's
5332 sticky bar. Writes through the pre-widget ``_pending_source_choice`` seam
5333 (assigning the picker's value inline is reconciled away by the browser), so
5334 it is safe as an ``on_click``.
5335 """
5336 st.session_state["_pending_source_choice"] = st.session_state.get(
5337 "_prev_source", DEMO_CHOICE
5338 )
5339 st.session_state["_show_upload_wizard"] = False
5340 st.session_state["setup_complete"] = True
5343def stay_in_wizard() -> None:
5344 """Dismiss BUG-31's leave prompt and carry on setting the dataset up.
5346 Records *which* view was declined rather than just clearing the prompt: the
5347 nav is still sitting on that view (nothing here navigates — see the note in
5348 ``main``), so a bare clear would re-raise the same question on the next
5349 rerun. Clicking a different view asks again, which is right.
5350 """
5351 st.session_state[WIZARD_STAY_KEY] = st.session_state.pop(WIZARD_LEAVE_KEY, None)
5354def discard_and_leave_wizard() -> None:
5355 """Abandon the half-built dataset and let the trip finish (BUG-31).
5357 :func:`leave_add_data_wizard` restores the source the wizard was opened
5358 *from*. For a **nav-triggered** prompt nothing here needs to navigate: the
5359 nav has been sitting on the requested view the whole time the prompt was
5360 up, so closing the wizard is all it takes for the next run to render it.
5362 **BUG-36 follow-up:** that assumption breaks for ✕ Cancel, whose prompt
5363 always names 🗂️ Data as the destination regardless of where the nav
5364 actually is — click Corpus Analysis, click Keep setting up, then Cancel, and
5365 the nav is still genuinely on Corpus Analysis throughout; closing the wizard alone
5366 left it there instead of on Data as promised. Requesting the recorded
5367 destination through the same ``main_nav`` seam :func:`url_state._go_data`
5368 and friends use is a no-op for the nav-triggered case (the router is
5369 already sitting on it, so this just re-affirms the same value) and is
5370 what actually moves it for Cancel's fixed one.
5371 """
5372 destination = st.session_state.pop(WIZARD_LEAVE_KEY, None)
5373 st.session_state.pop(WIZARD_STAY_KEY, None)
5374 leave_add_data_wizard()
5375 if destination:
5376 st.session_state["main_nav"] = destination
5379def render_data_source_picker(host=None) -> None:
5380 """Render the dataset picker and its + creation menu (UX-143).
5382 The menu sits between the dataset and trial selectors. Creation uses the
5383 existing manual editor or file wizard; the placeholder is picker-only.
5384 """
5385 from scanpath_studio.wizard import _enter_add_data_wizard
5387 entries = list(st.session_state.get("_data_source_entries") or [])
5388 if not entries:
5389 return
5390 kinds: dict[str, str] = dict(st.session_state.get("_data_source_kinds") or {})
5391 uploaded = list(st.session_state.get("_data_source_uploaded") or [])
5393 registry = public_dataset_registry()
5395 def _entry_label(token: str) -> str:
5396 # Reads the `registry` snapshot resolved just above rather than calling
5397 # `public_dataset_registry()` per token: the added corpora can change at
5398 # runtime, so one run must format its options against one
5399 # snapshot (which is also why the old `_public_dataset_label` helper,
5400 # which built its own, had no business being called from here — M6).
5401 tag = kinds.get(token, "")
5402 if token in registry:
5403 # `picker_name_for` is the single definition of what this list shows
5404 # — the entry's `short` plus, while DATA-27 is unfinished on main, a
5405 # (WIP) marker. Formatting only: the entry's key, its `short` and
5406 # its share slug are untouched, so dropping the marker later
5407 # invalidates no link and no saved config. Anything that tells the
5408 # user to "select X" reads it too, so the two cannot drift. The
5409 # snapshot is passed in for the M6 reason above it.
5410 name = _dataset_display_name(token, registry)
5411 elif token in uploaded:
5412 name = f"{_dataset_display_name(token, registry)} (yours)"
5413 else:
5414 name = _dataset_display_name(token, registry)
5415 return f"{tag} {name}".strip()
5417 # Keyed wrapper → stable `.st-key-…` selector for the spotlight tour.
5418 box = (host if host is not None else st).container(
5419 key="tour_grp_data_source",
5420 horizontal=True,
5421 vertical_alignment="bottom",
5422 gap="xsmall",
5423 wrap=False,
5424 )
5425 # Mirror the canonical key onto the widget key before it instantiates, so a
5426 # deep link / restore / wizard finalize shows up in the picker.
5427 current = st.session_state.get("data_source_choice")
5428 if current in entries:
5429 st.session_state["data_source_picker"] = current
5430 box.selectbox(
5431 "Select dataset",
5432 entries,
5433 format_func=_entry_label,
5434 key="data_source_picker",
5435 on_change=_on_data_source_pick,
5436 )
5437 # The menu opens as wide as the longest dataset name.
5438 widen_menu("data_source_picker", [_entry_label(entry) for entry in entries])
5439 # The help icon sits in the label row, right-aligned over +, not beside the
5440 # label: the bottom-aligned row keeps + level with the picker, so the icon
5441 # lands on the label's line.
5442 add_col = box.container(width="content", horizontal_alignment="right", gap=None)
5443 add_col.markdown(
5444 "",
5445 width="content",
5446 help=(
5447 "Which dataset the app is showing. Use + to create a scanpath or "
5448 "import files. Rename datasets, or remove the ones you added, on "
5449 f"the {ICONS['view_data']} Data Management page. More public "
5450 "datasets are planned."
5451 ),
5452 )
5453 # UX-200: named for screen readers; `styles.py` clips the name, so + is
5454 # still all that is drawn.
5455 with add_col.popover(
5456 "Add dataset",
5457 icon=ICONS["add"],
5458 help="Add dataset",
5459 wrap=True,
5460 key="add_dataset_menu",
5461 ):
5462 st.button(
5463 "Create manually",
5464 icon=ICONS["author"],
5465 key="add_manual_dataset_btn",
5466 help="Write a text and place its fixations, or resume your manual scanpath.",
5467 on_click=_enter_manual_dataset,
5468 width="stretch",
5469 )
5470 st.button(
5471 "Import files",
5472 icon=ICONS["upload"],
5473 key="import_dataset_btn",
5474 help="Add your Fixations and Words (interest areas) tables.",
5475 on_click=_enter_add_data_wizard,
5476 width="stretch",
5477 )
5480#: UX-174 — each dataset kind's word in the table, keyed by the picker's glyph,
5481#: and the Material Symbol the table draws beside it (the picker keeps its emoji,
5482#: since a selectbox option is plain text).
5483_DATASET_KIND_LABELS = {"🧪": "Demo", "✏️": "Manual", "🔒": "Private", "🌐": "Public"}
5484_DATASET_KIND_ICONS = {
5485 "Demo": ICONS["demo"],
5486 "Manual": ICONS["author"],
5487 "Private": ICONS["private"],
5488 "Public": ICONS["public"],
5489}
5491# Built-in and public dataset tokens are load-path identifiers, so changing
5492# them would break deep links and loader dispatch. Their table rename is a
5493# display alias; removing one hides it from this browser session. Uploaded
5494# datasets keep using the real store re-key/delete operations in `wizard.py`.
5495DATASET_ALIASES_KEY = "_dataset_display_aliases"
5496HIDDEN_DATASETS_KEY = "_hidden_dataset_tokens"
5499#: #374 — the built-in sources' names as shown. The tokens are a wire format
5500#: (links, saved state), so only the display changes.
5501_BUILTIN_DISPLAY_NAMES = {
5502 DEMO_CHOICE: "Bundled demo",
5503 MANUAL_SAMPLE_CHOICE: "Hand-drawn sample",
5504}
5507def _dataset_display_name(token: str, registry: dict | None = None) -> str:
5508 """User-facing dataset name without changing the source's stable token."""
5509 alias = (st.session_state.get(DATASET_ALIASES_KEY) or {}).get(token)
5510 if alias:
5511 return str(alias)
5512 if token == AUTHOR_CHOICE:
5513 return "My scanpath"
5514 if token in _BUILTIN_DISPLAY_NAMES:
5515 return _BUILTIN_DISPLAY_NAMES[token]
5516 registry = public_dataset_registry() if registry is None else registry
5517 return picker_name_for(token, registry) if token in registry else token
5520# `DATASET_COUNT_FIELDS` — the table's count columns and the only names a
5521# catalogue entry may publish under — lives in `dataset_table` (UX-174).
5524@dataclass(frozen=True)
5525class DatasetRowCounts:
5526 """What one row of the dataset table puts in its count columns (DATA-36).
5528 ``source`` says which of the two the row is showing — ``"loaded"`` (counted
5529 from rows this session holds) or ``"published"`` (the figures the corpus'
5530 own documentation, or a bundle manifest, states) — and it is one or the
5531 other, never a mixture: back-filling a measured row's gaps from the
5532 catalogue would make it read as one set of measurements while being two.
5534 ``differences`` is the check the whole item exists for. It holds every field
5535 both sides know and disagree on, as ``(published, loaded)``.
5536 """
5538 counts: Mapping[str, int | None]
5539 source: str
5540 differences: Mapping[str, tuple[int, int]]
5542 @property
5543 def exceeds_published(self) -> tuple[str, ...]:
5544 """Fields where **more** was loaded than the catalogue publishes.
5546 The one unambiguous signal that a published figure is wrong: a session
5547 cannot hold more of a corpus than the corpus has. The opposite — loading
5548 less — is the ordinary case (one OneStop regime, one part, a filtered
5549 export) and says nothing at all, which is why it is not flagged.
5550 """
5551 return tuple(
5552 field
5553 for field, (published, loaded) in self.differences.items()
5554 if loaded > published
5555 )
5558def dataset_row_counts(
5559 *,
5560 measured: Mapping[str, int | None] | None,
5561 published: Mapping[str, int] | None,
5562) -> DatasetRowCounts:
5563 """Resolve one row's counts from what was measured and what is published.
5565 ``measured`` is `remembered_dataset_counts`' answer, which is ``{}`` for a
5566 dataset the session has never held frames for and can carry ``None`` for a
5567 field that does not apply (no raw gaze, single-screen trials). A dict of
5568 nothing but ``None`` is *unknown*, not zero, and so does not count as a
5569 measurement.
5570 """
5571 measured = {k: v for k, v in (measured or {}).items() if v is not None}
5572 published = dict(published or {})
5573 if not measured:
5574 return DatasetRowCounts(published, "published" if published else "", {})
5575 differences = {
5576 field: (published[field], measured[field])
5577 for field in DATASET_COUNT_FIELDS
5578 if field in published
5579 and field in measured
5580 and published[field] != measured[field]
5581 }
5582 return DatasetRowCounts(measured, "loaded", differences)
5585def published_dataset_counts(token: str, registry: dict | None = None) -> dict:
5586 """The figures this catalogue publishes for a dataset, or ``{}``.
5588 Reached through `dataset_about`, so a public corpus, a packaged source and a
5589 prepared benchmark corpus all answer the same way — and an upload answers
5590 ``{}``, which is the honest answer: nothing here knows anything about it.
5591 """
5592 return dict(dataset_about(token, registry).get("published_counts") or {})
5595def benchmark_published_counts(entry) -> dict:
5596 """A prepared corpus' manifest counts, as dataset-table fields (DATA-36).
5598 The bundle already records `n_readers` / `n_texts` / `n_fixations` per
5599 corpus — the same numbers this table wants, one column each instead of the
5600 one sentence `_benchmark_size_caption` renders them as.
5602 A count `entry_count` cannot read comes back ``None``, and an absent one
5603 ``0``; **neither is published**. Publishing either would assert a corpus with
5604 no readers, which is exactly the overclaim `entry_count` warns against.
5605 """
5606 from scanpath_studio.eyegenbench import entry_count
5608 fields = (
5609 ("Participants", "n_readers"),
5610 ("Texts", "n_texts"),
5611 ("Fixations", "n_fixations"),
5612 )
5613 return {field: count for field, key in fields if (count := entry_count(entry, key))}
5616def _counts_store() -> dict:
5617 store = st.session_state.get(DATASET_COUNTS_STORE_KEY)
5618 if not isinstance(store, dict):
5619 store = {}
5620 st.session_state[DATASET_COUNTS_STORE_KEY] = store
5621 return store
5624def remembered_dataset_counts(
5625 token: str,
5626 words: pd.DataFrame | None,
5627 fixations: pd.DataFrame | None,
5628 raw_gaze: pd.DataFrame | None = None,
5629) -> dict:
5630 """This dataset's headline counts, computed at most once per version of it.
5632 **DATA-32.** Three cases, in order:
5634 1. **Frames in memory** (the open dataset, and every stored upload) — the
5635 counts are keyed on the frames' fingerprints, so a remembered entry is
5636 reused only while it still describes *these* rows. A remap, a re-upload
5637 or any other edit changes the fingerprint and the counts are recomputed;
5638 staleness is therefore not possible, which is what makes remembering them
5639 safe at all.
5640 2. **Frames not loaded, but counted before** — the remembered row is shown.
5641 This is the case the item is for: a public corpus you opened last week no
5642 longer costs minutes to list.
5643 3. **Never counted** — blank, as before. Nothing is read from disk to fill a
5644 table.
5646 The store is pruned by :func:`forget_dataset_counts` when a dataset leaves
5647 the session, and cleared with the recovery cache.
5648 """
5649 store = _counts_store()
5650 entry = store.get(token)
5651 if words is None and fixations is None and raw_gaze is None:
5652 counts = entry.get("counts") if isinstance(entry, dict) else None
5653 remembered = dict(counts) if isinstance(counts, dict) else {}
5654 # Recovery manifests written before this table matched the inspection
5655 # summary called the same value Readers. Preserve it without loading the
5656 # dataset merely to refresh a label.
5657 if "Participants" not in remembered and "Readers" in remembered:
5658 remembered["Participants"] = remembered.pop("Readers")
5659 return remembered
5660 key = [
5661 frame_fingerprint(words),
5662 frame_fingerprint(fixations),
5663 frame_fingerprint(raw_gaze),
5664 ]
5665 if isinstance(entry, dict) and entry.get("key") == key:
5666 return dict(entry.get("counts") or {})
5667 counts = _dataset_counts(words, fixations, raw_gaze, tuple(key))
5668 store[token] = {"key": key, "counts": dict(counts)}
5669 return dict(counts)
5672def forget_dataset_counts(keep: set | None = None) -> None:
5673 """Drop remembered counts for datasets that are no longer listed (DATA-32).
5675 ``keep=None`` forgets all of them — used by recovery-cache clearing and by
5676 the dataset-removal cache invalidation path.
5677 """
5678 store = _counts_store()
5679 for token in [t for t in store if keep is None or t not in keep]:
5680 store.pop(token, None)
5683@st.cache_data(show_spinner=False)
5684def _dataset_counts(
5685 _words: pd.DataFrame,
5686 _fixations: pd.DataFrame,
5687 _raw_gaze: pd.DataFrame,
5688 key,
5689) -> dict:
5690 """Cheap headline counts for every field in the dataset summary row.
5692 A few distinct-id unions and lengths — UX-54 asked for "measurements that
5693 are easy to calculate", and anything needing the measures pipeline would make
5694 *listing* the datasets as expensive as opening them. Cached on the frames'
5695 fingerprints (``key``), since this runs for every listed dataset on every
5696 rerun of the page.
5698 **``key`` has no leading underscore, and that is the whole point.**
5699 ``@st.cache_data`` skips underscore-prefixed arguments when it builds its
5700 cache key — that is how the frames are passed without being hashed — so the
5701 fingerprint argument was being skipped too (DATA-32 found it): the cache had
5702 exactly *one* entry, and every dataset after the first was served the first
5703 one's counts. UX-54 r2 had hidden it by counting only the open dataset.
5704 """
5706 words = _words if _words is not None else pd.DataFrame()
5707 fixations = _fixations if _fixations is not None else pd.DataFrame()
5708 raw_gaze = _raw_gaze if _raw_gaze is not None else pd.DataFrame()
5709 frames = (fixations, words, raw_gaze)
5711 def _union_unique(column: str):
5712 values = set()
5713 found = False
5714 for frame in frames:
5715 if frame is None or frame.empty or column not in frame.columns:
5716 continue
5717 found = True
5718 values.update(frame[column].dropna().astype(str).tolist())
5719 return len(values) if found else None
5721 # BUG-79: a count must not take the page down. `part_catalog` validates as
5722 # it counts and raises on screen metadata that disagrees across tables —
5723 # which is worth reporting where the figure is built, not by blanking the
5724 # whole 🗂️ Data page (and with it the way to switch to another dataset).
5725 try:
5726 screens = len(part_catalog(words, fixations, raw_gaze)) or None
5727 except ValueError as exc:
5728 logging.getLogger(__name__).warning("Screen count unavailable: %s", exc)
5729 screens = None
5730 # DATA-36: a trial is a **(participant, trial_id) pair** — the row the trial
5731 # picker lists, since `utils.build_combo_options` de-duplicates on exactly
5732 # that — not a distinct `trial_id`. The two coincide only where a corpus
5733 # numbers its trials globally. PoTeC names them after the text, so all 75
5734 # readers share the same twelve ids and counting ids reported **12 trials**
5735 # for a corpus whose own README says 900 (75 participants × 12 texts). It is
5736 # also the cheaper count: the union it replaces materialized every trial id
5737 # in the frame as a Python string (2.4 M of them on OneStop) to de-duplicate
5738 # what `drop_duplicates` had already reduced to a few thousand rows.
5739 trials = (
5740 len(trial_keys(words) | trial_keys(fixations) | trial_keys(raw_gaze)) or None
5741 )
5742 return {
5743 "Participants": _union_unique("participant_id"),
5744 # DATA-50: from every table that names a text, not the words alone.
5745 "Texts": len(text_ids(words, fixations, raw_gaze)) or None,
5746 "Trials": trials,
5747 "Screens": screens,
5748 "Words": len(words) or None,
5749 "Fixations": len(fixations) or None,
5750 "Gaze points": len(raw_gaze) or None,
5751 }
5754def _select_dataset(name: str) -> None:
5755 """Switch to a dataset from the UX-54 table, as the picker's callback does.
5757 Goes through the same ``_pending_source_choice`` seam the rename and the
5758 wizard finalize use: ``data_source_choice`` is a widget key elsewhere, so
5759 this is the one way an assignment lands before the widgets instantiate.
5760 """
5761 if name == AUTHOR_CHOICE:
5762 _remember_authoring_return()
5763 st.session_state["_pending_source_choice"] = name
5764 st.session_state["data_source_choice"] = name
5767#: UX-54 r2 — the upload the ✕ Delete button asked about, awaiting confirmation.
5768#: A plain session value, not a widget key: it is armed by a table callback and
5769#: read by the row of buttons the next run draws.
5770PENDING_DELETE_KEY = "_dataset_pending_delete"
5773def _dismiss_delete_confirmation() -> None:
5774 """``on_dismiss`` for the Remove dialog — the native ✕/Escape path.
5776 Without this, closing the dialog any way other than its own **Remove** /
5777 **Keep it** buttons left `PENDING_DELETE_KEY` armed: the fragment reopened
5778 the same confirmation on its very next rerun, however that rerun was
5779 triggered, and — since only one Streamlit dialog can show at a time — it
5780 then also *shadowed* any other row action clicked afterward. Same bug as
5781 ``_dismiss_dataset_about`` below, same fix.
5782 """
5783 st.session_state.pop(PENDING_DELETE_KEY, None)
5786@st.dialog("Remove this dataset?", on_dismiss=_dismiss_delete_confirmation)
5787@guarded()
5788def _delete_confirmation_dialog(
5789 token: str, *, uploaded: set[str], available: list[str]
5790) -> None:
5791 """The modal body — UX-79. Opened by ``_render_delete_confirmation``.
5793 **BUG-36:** handled by the button's *return value*, not ``on_click`` — an
5794 ``st.dialog`` body is a fragment, so an ``on_click`` callback here reran
5795 only the dialog: the deletion happened, but ``main()`` never re-executed,
5796 so the modal sat there looking inert. ``st.rerun(scope="app")`` both closes
5797 the modal and re-renders the page underneath (see ``tour.py``'s
5798 ``_tutorial_library_dialog``, which hit the same trap first).
5799 """
5800 owned = token in uploaded
5801 if owned:
5802 # BUG-95: said "and annotations" while removing none. The count is the
5803 # dataset's own annotations, which since DATA-48 it alone holds.
5804 from scanpath_studio.wizard import upload_annotations
5806 count = len(upload_annotations(token))
5807 taken = (
5808 f", column mapping and its {count:,} annotation{'' if count == 1 else 's'}"
5809 if count
5810 else " and column mapping"
5811 )
5812 st.warning(
5813 f"Remove **{_dataset_display_name(token)}**? Its tables{taken} are "
5814 "deleted here and from this computer's saved copy — no undo."
5815 )
5816 if count:
5817 st.caption(
5818 "To keep the annotations, open it and **Export** them from "
5819 "**Annotations** first."
5820 )
5821 else:
5822 st.caption(
5823 f"Remove **{_dataset_display_name(token)}** from the list of datasets "
5824 "for this session? The packaged or public source data is not deleted."
5825 )
5826 remaining = [entry for entry in available if entry != token]
5827 yes, no = st.columns(2)
5828 if yes.button(
5829 "Remove",
5830 key="dataset_delete_confirm",
5831 type="primary",
5832 width="stretch",
5833 disabled=not remaining,
5834 ):
5835 pending = st.session_state.pop(PENDING_DELETE_KEY, None)
5836 if owned:
5837 # Local import, like `_enter_add_data_wizard` above: `wizard`
5838 # imports `app` back, so it cannot be imported at module load.
5839 from scanpath_studio.wizard import _remove_dataset
5841 _remove_dataset(pending)
5842 else:
5843 hidden = set(st.session_state.get(HIDDEN_DATASETS_KEY) or [])
5844 hidden.add(pending)
5845 st.session_state[HIDDEN_DATASETS_KEY] = sorted(hidden)
5846 if st.session_state.get("data_source_choice") == pending:
5847 st.session_state["_pending_source_choice"] = remaining[0]
5848 st.rerun(scope="app")
5849 if not remaining:
5850 st.caption("At least one dataset must remain available.")
5851 if no.button(
5852 "Keep it",
5853 key="dataset_delete_cancel",
5854 width="stretch",
5855 ):
5856 st.session_state.pop(PENDING_DELETE_KEY, None)
5857 st.rerun(scope="app")
5860def _render_delete_confirmation(host, tokens: list, uploaded: set[str]) -> None:
5861 """The confirm step between ✕ Delete and the dataset actually going away.
5863 Deleting an upload drops its frames, its mapping and its annotations
5864 (BUG-95, DATA-48) from the session with no undo, and the button that starts it sits on a row that
5865 opens the dataset when clicked anywhere else — so the click arms this, and
5866 this asks.
5868 **UX-79** made it a modal rather than a block under the table: the question
5869 is raised by a click *in* the table, and on a long list of datasets a
5870 container below it can be off-screen from the row that asked. The arming
5871 flag is unchanged — a dialog is opened by calling it, so what moved is where
5872 the flag is read. A token that has since disappeared (the dataset was
5873 removed another way) disarms itself instead of opening a dialog about
5874 nothing.
5875 """
5876 token = st.session_state.get(PENDING_DELETE_KEY)
5877 if token is None:
5878 return
5879 if token not in tokens:
5880 st.session_state.pop(PENDING_DELETE_KEY, None)
5881 return
5882 _delete_confirmation_dialog(token, uploaded=uploaded, available=tokens)
5885def _unique_dataset_alias(requested: str, token: str, tokens: list[str]) -> str:
5886 """A non-empty display name that does not duplicate another table row."""
5887 base = requested.strip() or _dataset_display_name(token)
5888 used = {
5889 _dataset_display_name(other).casefold() for other in tokens if other != token
5890 }
5891 candidate = base
5892 suffix = 2
5893 while candidate.casefold() in used:
5894 candidate = f"{base} ({suffix})"
5895 suffix += 1
5896 return candidate
5899def _overview_sentence(text: str) -> tuple[str, str]:
5900 """Split a dataset description into its opening sentence and the rest.
5902 UX-137: the 🗂️ Data page shows the first sentence as the overview under
5903 "What's in this dataset". UX-177 made every catalogue description one
5904 sentence, so this only trims a prepared benchmark corpus', whose tail is
5905 its geometry, license and citation.
5907 A boundary inside a `code span` does not count. Several descriptions end on
5908 a module path (``python -m scanpath_studio.update_sample_data``), and cutting
5909 one at its first dot reads as a typo rather than as a summary.
5910 """
5911 body = str(text or "").strip()
5912 in_code = False
5913 for index, char in enumerate(body):
5914 if char == "`":
5915 in_code = not in_code
5916 continue
5917 if char not in ".!?" or in_code:
5918 continue
5919 rest = body[index + 1 :]
5920 if not rest.strip():
5921 break
5922 # A real boundary: whitespace, then something that starts a sentence.
5923 if rest[:1].isspace() and rest.lstrip()[:1].isupper():
5924 return body[: index + 1], rest.strip()
5925 return body, ""
5928def dataset_description(token: str, registry: dict | None = None) -> tuple[str, bool]:
5929 """The dataset's description, and whether the user wrote it.
5931 The user's own text (`DATASET_DESCRIPTIONS_KEY`, from the add wizard or
5932 ✏️ Edit dataset) wins, and the catalogue's (`dataset_about`) is the
5933 fallback. An empty string the user saved counts as theirs: they cleared it.
5934 """
5935 own = st.session_state.get(DATASET_DESCRIPTIONS_KEY) or {}
5936 if token in own:
5937 return str(own[token]), True
5938 return str(dataset_about(token, registry).get("description") or ""), False
5941def set_dataset_description(token: str, text: str) -> None:
5942 """Store ``text`` as the dataset's own description."""
5943 own = dict(st.session_state.get(DATASET_DESCRIPTIONS_KEY) or {})
5944 own[token] = str(text or "").strip()
5945 st.session_state[DATASET_DESCRIPTIONS_KEY] = own
5948def _description_field_key(token: str) -> str:
5949 return f"dataset_description_{_dataset_row_slug(token)}"
5952def _description_draft(token: str) -> str | None:
5953 """The description typed on the open editor, if it differs from the saved
5954 one (``None`` when it does not, or the field has not drawn)."""
5955 key = _description_field_key(token)
5956 if key not in st.session_state:
5957 return None
5958 text = str(st.session_state.get(key) or "").strip()
5959 return None if text == dataset_description(token)[0].strip() else text
5962def render_description_field(host, token: str) -> None:
5963 """✏️ Edit dataset's **Description** — the sentence under its name.
5965 Held, like everything else on the screen, until ✅ Save changes
5966 (`commit_editor_staging`); ✕ Cancel drops it with the rest of the edit.
5967 The catalogue's own text is only adopted as the user's when they change it.
5968 """
5969 key = _description_field_key(token)
5970 if key not in st.session_state:
5971 st.session_state[key] = dataset_description(token)[0]
5972 host.text_area(
5973 "Description",
5974 key=key,
5975 placeholder="What this dataset is — the participants, the texts, the language.",
5976 help=f"Shown under the dataset's name on the {ICONS['view_data']} Data "
5977 f"Management page. Saved with **{ICONS['confirm']} Save changes**.",
5978 height=80,
5979 # Streamlit 1.65: read only by Save changes / Cancel, so typing needs
5980 # no rerun.
5981 on_change="ignore",
5982 # A draft outlives a visit to another view while the editor is open.
5983 persist_state="session",
5984 )
5987def _render_dataset_overview(token: str, *, registry: dict) -> None:
5988 """The open dataset in a sentence, and its home page.
5990 UX-177 cut this to what we stand behind and a new reader can use: the
5991 description (its first sentence, unless the user wrote it), the corpus'
5992 home page on the same line, and — only where it changes how a figure is
5993 read — one ``reading_note`` (PoTeC's reconstructed positions, the demo's
5994 synthesized raw gaze). The ❔ *About this dataset* popover it replaces
5995 carried maintainer provenance: how each published figure was counted, a
5996 published-vs-loaded table, and a coordinate badge for every dataset. The
5997 table's Status already says whether its numbers are published or loaded.
5999 Editing it is ✏️ **Edit dataset**, under the overview (UX-178).
6000 """
6001 about = dataset_about(token, registry)
6002 text, own = dataset_description(token, registry)
6003 overview = text if own else _overview_sentence(text)[0]
6004 line = st.container(
6005 key="dataset_overview_line",
6006 horizontal=True,
6007 vertical_alignment="center",
6008 gap="small",
6009 )
6010 if link := about.get("link"):
6011 overview = f"{overview} [Home page ↗]({link})".strip()
6012 if overview:
6013 line.caption(overview, width="content")
6014 if note := about.get("reading_note"):
6015 st.caption(f"{ICONS['info']} {note}")
6018@st.cache_data(show_spinner=False)
6019def _c_annotation_trials(_combos: pd.DataFrame, key) -> frozenset[tuple[str, str]]:
6020 """The ``(participant, trial)`` pairs of ``_combos``, as strings.
6022 Cached on the frame's fingerprint (``key``): the Data page asks on every
6023 rerun — a row ticked in the Annotations table is one — and the answer only
6024 changes with the dataset.
6025 """
6026 if not {"participant_id", "trial_id"} <= set(_combos.columns):
6027 return frozenset()
6028 pairs = _combos[["participant_id", "trial_id"]].drop_duplicates()
6029 return frozenset(
6030 zip(pairs["participant_id"].astype(str), pairs["trial_id"].astype(str))
6031 )
6034def _annotation_trials(combos: pd.DataFrame | None) -> frozenset[tuple[str, str]]:
6035 """The open dataset's ``(participant, trial)`` pairs, for its Annotations tab."""
6036 if combos is None or combos.empty:
6037 return frozenset()
6038 return _c_annotation_trials(combos, frame_fingerprint(combos))
6041@st.cache_data(show_spinner=False)
6042def _c_annotation_trial_labels(
6043 _combos: pd.DataFrame, key, composite_cols: tuple[str, ...]
6044) -> dict[str, str]:
6045 """Each trial as the trial picker writes it, cached on the frame's
6046 fingerprint (``key``): a Python loop over every trial, asked on every rerun
6047 of the Data page."""
6048 return trial_id_layout(_combos, composite_cols=composite_cols)[0]
6051def _annotation_trial_labels(combos: pd.DataFrame | None) -> dict[str, str] | None:
6052 """The open dataset's trial labels, for its Annotations tab (#374 F5)."""
6053 if combos is None:
6054 return None
6055 composite = tuple(st.session_state.get("_composite_trial_columns") or ())
6056 return _c_annotation_trial_labels(combos, frame_fingerprint(combos), composite)
6059def render_dataset_inspection_head(token: str) -> None:
6060 """*What's in the `<name>` dataset* and the overview under it.
6062 ✏️ **Edit dataset** is not on this line: it is drawn by
6063 :func:`render_dataset_edit_button`, below the description and checks and
6064 above the inspection subtabs.
6065 """
6066 # #374 F30: the name alone — "the `Dataset 1` dataset" said it twice.
6067 label = _dataset_display_name(token).replace("*", r"\*")
6068 st.subheader(f"{ICONS['search']} What's in **{label}**")
6069 _render_dataset_overview(token, registry=public_dataset_registry())
6072def render_dataset_edit_button(token: str) -> None:
6073 """✏️ **Edit dataset**, between the dataset's overview and its subtabs.
6075 UX-178: the section's one action is a button of its own, not a link-weight
6076 one beside the description, so it reads as editing the whole dataset — its
6077 name, its description and its setup, all on the screen it opens. It does
6078 not apply to the add-dataset wizard's pending dataset or to the authoring
6079 canvas, which are not rows of the table.
6080 """
6081 if token in (UPLOAD_CHOICE, AUTHOR_CHOICE):
6082 return
6083 st.button(
6084 "Edit dataset",
6085 icon=ICONS["edit"],
6086 key="dataset_edit_btn",
6087 on_click=_edit_open_dataset,
6088 args=(token,),
6089 help="Open the authoring editor — change the text, drag fixations, "
6090 "or edit their timing."
6091 if token == MANUAL_SAMPLE_CHOICE
6092 else "Its name, description, column mapping, recording setup, "
6093 "location and metadata tables.",
6094 )
6097def _builtin_name_draft(token: str) -> str | None:
6098 """The name typed for a dataset that is not an upload, if it is a new one."""
6099 if EDITOR_NAME_FIELD_KEY not in st.session_state:
6100 return None
6101 requested = str(st.session_state.get(EDITOR_NAME_FIELD_KEY) or "").strip()
6102 if not requested or requested == _dataset_display_name(token):
6103 return None
6104 return requested
6107def _apply_builtin_name(token: str) -> None:
6108 """✅ Save changes' rename of a dataset that is not an upload.
6110 A built-in or public source's token is a load-path identifier (deep links,
6111 loader dispatch), so its name is a display alias — nothing to re-key.
6112 """
6113 requested = _builtin_name_draft(token)
6114 if requested is None:
6115 return
6116 tokens = list(st.session_state.get("_data_source_entries") or [])
6117 final = _unique_dataset_alias(requested, token, tokens)
6118 aliases = dict(st.session_state.get(DATASET_ALIASES_KEY) or {})
6119 aliases[token] = final
6120 st.session_state[DATASET_ALIASES_KEY] = aliases
6123def _stage_upload_name() -> None:
6124 """``on_change`` of **Name** for an upload: hold it for ✅ Save changes.
6126 Kept in a plain ``_remap_`` key rather than read back off the widget, so it
6127 survives whatever the widget's own state does between runs, and is swept
6128 with the rest of the edit on Cancel or Save.
6129 """
6130 st.session_state[EDITOR_PENDING_NAME_KEY] = str(
6131 st.session_state.get(EDITOR_NAME_FIELD_KEY) or ""
6132 ).strip()
6135def render_name_field(host, token: str) -> None:
6136 """✏️ Edit dataset's **Name** (UX-178; renaming used to be a dialog).
6138 Applied by ✅ Save changes with the rest of the edit, and an unsaved change
6139 until then. An upload's name is the key its every editor widget is filed
6140 under, so `tabs._apply_remap` re-keys it last; any other dataset's name is
6141 a display alias (`_apply_builtin_name`).
6142 """
6143 uploaded = token in (st.session_state.get("_datasets") or {})
6144 if EDITOR_NAME_FIELD_KEY not in st.session_state:
6145 st.session_state[EDITOR_NAME_FIELD_KEY] = st.session_state.get(
6146 EDITOR_PENDING_NAME_KEY
6147 ) or _dataset_display_name(token)
6148 host.text_input(
6149 "Name",
6150 key=EDITOR_NAME_FIELD_KEY,
6151 required=True,
6152 # An upload stages its name on change; a built-in's is read only by
6153 # Save changes / Cancel, so it needs no rerun (Streamlit 1.65).
6154 on_change=_stage_upload_name if uploaded else "ignore",
6155 help="Shown in the list of datasets and the dataset picker. Saved with "
6156 f"**{ICONS['confirm']} Save changes**.",
6157 # A draft outlives a visit to another view while the editor is open.
6158 persist_state="session",
6159 )
6162def _open_mapping_editor() -> None:
6163 """Raise ✏️ Edit dataset on the open dataset, focused on its mapping."""
6164 token = str(st.session_state.get("data_source_choice") or "")
6165 if token:
6166 st.session_state[FOCUS_MAPPING_KEY] = token
6167 st.session_state[DATASET_EDITOR_OPEN_KEY] = True
6168 st.session_state[_EDITOR_SCROLL_KEY] = True
6171def dataset_added_message(
6172 name: str, words: pd.DataFrame, fixations: pd.DataFrame
6173) -> str:
6174 """The toast ✅ Add dataset ends with (#374 F30): "**Dataset 1** added —
6175 24 trials, 2 participants." The counts are left out when there are none
6176 (a raw-gaze-only dataset counts its trials elsewhere)."""
6177 readers: set = set()
6178 for frame in (words, fixations):
6179 if frame is not None and "participant_id" in frame.columns:
6180 readers |= set(frame["participant_id"].dropna().astype(str))
6181 readers.discard("") # a stimulus-level word table's placeholder
6182 trials = count_trials(words, fixations)
6183 shown = name.replace("*", r"\*")
6184 head = f"**{shown}** added"
6185 if not trials:
6186 return f"{head}."
6187 return f"{head} — {plural(trials, 'trial')}, {plural(len(readers), 'participant')}."
6190@st.dialog(f"{ICONS['warning']} Check the Trial ID mapping")
6191@guarded()
6192def _trial_identity_alert_dialog(asked_by: str, warning: str) -> None:
6193 """VAL-9 — VAL-7's verdict, raised where the Trial ID was just chosen.
6195 Opened by ``main`` on the run *after* ✅ Add dataset or ✅ Save changes, from
6196 the report those buttons' frames produced — so the check runs on the finished
6197 dataset (it is the sampled one, PERF-6, which is what keeps it cheap enough
6198 to sit in the commit path at all) rather than as a permanent page-wide
6199 banner about a decision made twice.
6201 Two answers, and both are real: **keep it** for a corpus where several
6202 readings of one text genuinely are one trial, and **edit the mapping** for
6203 the far commoner case where a column was simply left out of the Trial ID.
6204 Buttons are handled by their return value, never ``on_click`` — a dialog
6205 body is a fragment (see ``_leave_dataset_editor_dialog``).
6206 """
6207 st.warning(warning, icon=ICONS["warning"])
6208 st.caption(
6209 "A Trial ID that doesn't fully identify one reading concatenates several "
6210 "into one scanpath — which renders perfectly happily, as an ordinary "
6211 "scanpath with a lot of regressions. The full evidence is under "
6212 f"{ICONS['view_data']} Data Management → **Edit dataset** → **4 · Trial "
6213 "identity**."
6214 )
6215 edit_col, keep_col = st.columns(2, gap="small")
6216 if edit_col.button(
6217 f"{ICONS['edit']} Edit the mapping",
6218 key="trial_identity_alert_edit",
6219 type="primary",
6220 width="stretch",
6221 help=f"Open {ICONS['edit']} Edit dataset on the Trial ID mapping this verdict is about.",
6222 ):
6223 _open_mapping_editor()
6224 st.rerun(scope="app")
6225 if keep_col.button(
6226 "Keep it as is",
6227 key="trial_identity_alert_keep",
6228 width="stretch",
6229 help=f"Dismiss. Nothing changes; the verdict stays under {ICONS['view_data']} "
6230 "Data Management → Edit dataset → 4 · Trial identity.",
6231 ):
6232 st.rerun(scope="app")
6233 if asked_by == "add":
6234 st.caption(
6235 "Checked automatically because the dataset was just added. It is "
6236 f"already on the {ICONS['view_data']} Data Management page's list either way."
6237 )
6240#: UX-106 — whether the editor's ✕ Cancel confirmation is open. A pending
6241#: flag rather than a button return value, the house pattern: a callback may not
6242#: open a dialog, and a dialog opened from a return value is lost on the next
6243#: full rerun.
6244_EDITOR_LEAVE_PENDING_KEY = "_dataset_editor_leave_pending"
6245#: UX-197 — the dataset a row click asked for while the editor had unsaved
6246#: changes. The table stays on screen above the editor now, so a click on
6247#: another row is a way out of the editor too, and goes through the same
6248#: confirmation; ✕ Leave then opens this dataset.
6249_EDITOR_LEAVE_TARGET_KEY = "_dataset_editor_leave_target"
6250#: UX-197 — set by whatever opens the editor, popped by its bar: the editor
6251#: opens under the table, so the page is brought down to it once.
6252_EDITOR_SCROLL_KEY = "_dataset_editor_scroll"
6254#: Bring the editor's top under the app's header — in its own scroller, never
6255#: by `scrollIntoView` (which moves the document; see `tour.py`). Waits while
6256#: Streamlit is still laying the editor out, then keeps the editor's top in
6257#: place for a second, because the page above it is still settling and a
6258#: single scroll lands wherever the layout happened to be at that moment.
6259_SCROLL_TO_EDITOR_SCRIPT = """<script>
6260(function () {
6261 const doc = window.parent.document;
6262 const win = doc.defaultView;
6263 let tries = 0, aligned = 0;
6264 (function attempt() {
6265 const el = doc.querySelector(".st-key-data_dataset_editor");
6266 const r = el && el.getBoundingClientRect();
6267 if (!r || r.height === 0) {
6268 if (++tries < 20) setTimeout(attempt, 150);
6269 return;
6270 }
6271 for (let box = el.parentElement; box; box = box.parentElement) {
6272 const cs = win.getComputedStyle(box);
6273 if (/(auto|scroll|overlay)/.test(cs.overflowY)
6274 && box.scrollHeight > box.clientHeight + 4) {
6275 const b = box.getBoundingClientRect();
6276 box.scrollTop += r.top - b.top - 56;
6277 break;
6278 }
6279 }
6280 if (++aligned < 8) setTimeout(attempt, 150);
6281 })();
6282})();
6283</script>"""
6286#: ✏️ Edit dataset's record of the metadata tables as the edit found them.
6287#: The three metadata sections are the add screen's, and attach a table as its
6288#: file is read — so rather than stage them, the editor notes what was attached
6289#: when it opened, ✕ Cancel puts that back, and ✅ Save changes keeps what is
6290#: there. Taken by `hold_editor_staging` on the editor's first run.
6291_EDITOR_SNAPSHOT_KEY = "_dataset_editor_snapshot"
6292#: ✕ Cancel's metadata restore, parked for the next run to apply before the
6293#: metadata sections draw (`apply_editor_restore`) — the Leave confirmation is
6294#: a dialog, whose click runs after the page's widgets.
6295_EDITOR_RESTORE_KEY = "_dataset_editor_restore"
6298def _metadata_grain_state(grain: str) -> dict:
6299 """One metadata grain's attached table and the read behind it."""
6300 key, raw, file = metadata_mod.grain_keys(grain)
6301 table = st.session_state.get(key)
6302 frame = getattr(table, "frame", None)
6303 # Its content, not its identity: the sections rebuild the table every run.
6304 # Small (one row per reader, trial or text), so hashing it is cheap.
6305 digest = None
6306 if isinstance(frame, pd.DataFrame):
6307 try:
6308 cells = pd.util.hash_pandas_object(frame, index=False).to_numpy().tobytes()
6309 except (TypeError, ValueError): # unhashable cells — hash their text
6310 cells = frame.to_csv(index=False).encode("utf-8")
6311 digest = (tuple(map(str, frame.columns)), hashlib.sha256(cells).hexdigest())
6312 return {
6313 "table": table,
6314 "raw": st.session_state.get(raw),
6315 "file": st.session_state.get(file),
6316 "name": st.session_state.get(f"_{grain}_metadata_name"),
6317 "content": digest,
6318 }
6321def _metadata_grains_changed(snapshot: dict) -> list[str]:
6322 """The grains whose attached table differs from the editor's snapshot."""
6323 changed = []
6324 for grain, before in (snapshot.get("grains") or {}).items():
6325 now = _metadata_grain_state(grain)
6326 if (
6327 (now["table"] is None) != (before["table"] is None)
6328 or now["raw"] is not before["raw"]
6329 or now["file"] != before["file"]
6330 or now["content"] != before["content"]
6331 ):
6332 changed.append(grain)
6333 return changed
6336def hold_editor_staging(token: str) -> None:
6337 """Note the metadata tables as this edit finds them (once per edit)."""
6338 held = st.session_state.get(_EDITOR_SNAPSHOT_KEY)
6339 if isinstance(held, dict) and held.get("token") == token:
6340 return
6341 st.session_state[_EDITOR_SNAPSHOT_KEY] = {
6342 "token": token,
6343 "owner": st.session_state.get(metadata_mod.OWNER_KEY),
6344 "grains": {
6345 grain: _metadata_grain_state(grain)
6346 for grain in (metadata_mod.GRAIN_PARTICIPANT, "trial", "text")
6347 },
6348 }
6351def editor_staging_dirty() -> bool:
6352 """Whether the open editor's name, description or metadata tables differ
6353 from what it opened on — the part of an edit `tabs.dataset_editor_is_dirty`
6354 does not see."""
6355 snapshot = st.session_state.get(_EDITOR_SNAPSHOT_KEY)
6356 if not isinstance(snapshot, dict):
6357 return False
6358 token = str(snapshot.get("token") or "")
6359 return bool(
6360 _description_draft(token) is not None
6361 or _builtin_name_draft(token) is not None
6362 or _metadata_grains_changed(snapshot)
6363 )
6366def _editor_is_dirty() -> bool:
6367 """Whether ✕ Cancel would lose anything (UX-107): the mapping, setup and
6368 uploads (`tabs.dataset_editor_is_dirty`), or the name, description and
6369 metadata tables."""
6370 return dataset_editor_is_dirty() or editor_staging_dirty()
6373def commit_editor_staging(token: str) -> None:
6374 """✅ Save changes' share of the edit: the description, a built-in's name,
6375 and the metadata tables as they now stand.
6377 Called by both Saves — an upload's (`tabs._apply_remap`, before it re-keys
6378 the dataset under a new name) and a built-in's — once they know the save
6379 goes ahead.
6380 """
6381 if (text := _description_draft(token)) is not None:
6382 set_dataset_description(token, text)
6383 if token not in (st.session_state.get("_datasets") or {}):
6384 _apply_builtin_name(token)
6385 _drop_description_drafts()
6386 # Kept, not restored: what is attached now is what was saved.
6387 st.session_state.pop(_EDITOR_SNAPSHOT_KEY, None)
6390def _drop_description_drafts() -> None:
6391 for key in [
6392 k
6393 for k in list(st.session_state)
6394 if isinstance(k, str) and k.startswith("dataset_description_")
6395 ]:
6396 st.session_state.pop(key, None)
6399def apply_editor_restore() -> None:
6400 """Put back the metadata tables a cancelled edit changed (`_EDITOR_RESTORE_KEY`).
6402 Runs before `metadata.activate_dataset` and before the sections draw, so
6403 the tables go back to the dataset they were taken from. A table the edit
6404 replaced or detached returns as a *restored* one — its file is no longer in
6405 the uploader, so it is re-attached the way the recovery cache re-attaches
6406 a table (`metadata.mark_restored`) rather than read again.
6407 """
6408 snapshot = st.session_state.pop(_EDITOR_RESTORE_KEY, None)
6409 if not isinstance(snapshot, dict):
6410 return
6411 if snapshot.get("owner") != st.session_state.get(metadata_mod.OWNER_KEY):
6412 return
6413 changed = _metadata_grains_changed(snapshot)
6414 if changed:
6415 # The uploader still holds the cancelled edit's file in the browser.
6416 metadata_mod.reset_uploads(st.session_state)
6417 for grain in changed:
6418 before = snapshot["grains"][grain]
6419 key, raw, file = metadata_mod.grain_keys(grain)
6420 for name in (
6421 f"{grain}_metadata_id_column",
6422 f"{grain}_metadata_keep_fields",
6423 ):
6424 st.session_state.pop(name, None)
6425 if before["table"] is None:
6426 for name in (key, raw, file, f"_{grain}_metadata_name"):
6427 st.session_state.pop(name, None)
6428 continue
6429 metadata_mod.mark_restored(st.session_state, grain, before["table"])
6430 if before["name"] is not None:
6431 st.session_state[f"_{grain}_metadata_name"] = before["name"]
6434def _close_dataset_editor() -> None:
6435 """``on_click`` for the editor's way out — back to 📂 Available datasets."""
6436 st.session_state.pop(DATASET_EDITOR_OPEN_KEY, None)
6437 st.session_state.pop(FOCUS_MAPPING_KEY, None)
6438 st.session_state.pop(_EDITOR_LEAVE_PENDING_KEY, None)
6439 st.session_state.pop(_EDITOR_LEAVE_TARGET_KEY, None)
6440 # Anything typed into the editor and not saved goes with it — including a
6441 # table uploaded to fill a missing half, which is only a *pending* attach
6442 # until ✅ Save changes runs.
6443 for key in [k for k in st.session_state if str(k).startswith("_remap_")]:
6444 st.session_state.pop(key, None)
6445 # …and the mapping widgets' own answers, which outlive the screen.
6446 discard_editor_widgets()
6447 st.session_state.pop(EDITOR_NAME_FIELD_KEY, None)
6448 # DATA-46: "use the current estimate" is a choice for one editing session.
6449 for key in [k for k in st.session_state if str(k).endswith("_setup_reestimate")]:
6450 st.session_state.pop(key, None)
6451 # A built-in source's unsaved mapping goes too (✅ Save changes has already
6452 # dropped what it would restore).
6453 _discard_builtin_mapping_edit()
6454 # The name and description typed into it, and the metadata tables it
6455 # changed (✅ Save changes has already dropped the snapshot it would
6456 # restore them from).
6457 _discard_editor_staging()
6460def _discard_editor_staging() -> None:
6461 """Drop the editor's description drafts; park its metadata restore."""
6462 _drop_description_drafts()
6463 snapshot = st.session_state.pop(_EDITOR_SNAPSHOT_KEY, None)
6464 if isinstance(snapshot, dict) and _metadata_grains_changed(snapshot):
6465 st.session_state[_EDITOR_RESTORE_KEY] = snapshot
6468def _ask_leave_dataset_editor() -> None:
6469 """✕ Cancel: confirm only when there is something to lose (UX-106/UX-107).
6471 An editor nobody has typed into is a screen you are simply leaving, and a
6472 modal asking whether you are sure is friction for a decision with no
6473 consequence. `tabs.dataset_editor_is_dirty` answers the question and errs
6474 towards *dirty*, so the confirmation is skipped only when the mapping, the
6475 recording setup and the uploads are all exactly as the editor opened.
6476 """
6477 if not _editor_is_dirty():
6478 _close_dataset_editor()
6479 return
6480 st.session_state[_EDITOR_LEAVE_PENDING_KEY] = True
6483def _dismiss_leave_dataset_editor() -> None:
6484 st.session_state.pop(_EDITOR_LEAVE_PENDING_KEY, None)
6485 st.session_state.pop(_EDITOR_LEAVE_TARGET_KEY, None)
6488@st.dialog("Leave without saving?", on_dismiss=_dismiss_leave_dataset_editor)
6489@guarded()
6490def _leave_dataset_editor_dialog() -> None:
6491 """Confirm discarding an in-progress edit — the add screen's ✕ Cancel.
6493 The add screen has always confirmed (`_render_leave_prompt`), because
6494 leaving it throws away an upload. This screen used to leave straight away
6495 on the reasoning that "an edit is applied by its own button, so leaving
6496 discards nothing" — which stopped being true once a *table* could be
6497 uploaded here and be waiting on Save.
6499 **BUG-49 — both buttons are handled by their return value, never by
6500 ``on_click``.**
6501 A dialog body is a fragment, so a callback in here reruns *the dialog* — ✕
6502 Leave popped the editor's open flag and then sat there with the modal still
6503 up, and the next click ("Keep editing") ran the whole-app rerun that finally
6504 acted on it. The screen therefore left on *Keep editing* and did nothing on
6505 *Leave*. Same lesson as ``_delete_confirmation_dialog``.
6506 """
6507 st.caption(
6508 "Changes you have already saved are kept. Anything edited since — the "
6509 "name and description, the mapping, the recording setup, the metadata "
6510 "tables, and any table uploaded to fill a missing one — is discarded."
6511 )
6512 leave, stay = st.columns(2, gap="small")
6513 if leave.button(
6514 "✕ Leave",
6515 key="dataset_editor_leave_confirm",
6516 type="primary",
6517 width="stretch",
6518 ):
6519 target = st.session_state.get(_EDITOR_LEAVE_TARGET_KEY)
6520 _close_dataset_editor()
6521 if target:
6522 # Through the pre-widget seam only: this is a button's return
6523 # value, after the picker has instantiated in this run.
6524 st.session_state["_pending_source_choice"] = target
6525 st.rerun(scope="app")
6526 if stay.button("Keep editing", key="dataset_editor_leave_cancel", width="stretch"):
6527 _dismiss_leave_dataset_editor()
6528 st.rerun(scope="app")
6531def _render_dataset_editor_bar(host, data_choice: str) -> None:
6532 """DATA-35 — the ✏️ Edit dataset screen's sticky header.
6534 Deliberately the add-dataset screen's bar, down to the CSS class: the ask
6535 was that "the Add Dataset and Edit Dataset screens should be very similar",
6536 and the two screens ask the same questions of the same dataset — one before
6537 it exists and one after. The only difference left is the title, which names
6538 the dataset; UX-106 made the way out the add screen's ✕ Cancel, with the
6539 same confirmation, because a table uploaded here to fill a missing half is
6540 pending until ✅ Save changes runs and leaving would drop it in silence.
6541 """
6542 name = _dataset_display_name(
6543 str(st.session_state.get("data_source_choice") or data_choice)
6544 )
6545 bar = host.container(key="dataset_editor_bar")
6546 title_col, back_col = bar.columns([8, 2], vertical_alignment="center")
6547 title_col.markdown(
6548 f'<div class="sps-wiz-title">{icon_html("edit")} Edit {html.escape(name)}</div>',
6549 unsafe_allow_html=True,
6550 )
6551 back_col.button(
6552 # UX-106: the add screen's ✕ Cancel, in the same filled blue and the
6553 # same corner — the two screens are the same screen, before and after.
6554 "✕ Cancel",
6555 key="dataset_editor_close",
6556 type="primary",
6557 on_click=_ask_leave_dataset_editor,
6558 width="stretch",
6559 help="Leave the editor. Changes you have already saved are kept; "
6560 "anything unsaved is discarded, after a confirmation.",
6561 )
6562 if st.session_state.get(_EDITOR_LEAVE_PENDING_KEY):
6563 _leave_dataset_editor_dialog()
6564 if st.session_state.pop(_EDITOR_SCROLL_KEY, False):
6565 with bar:
6566 embed_html_iframe(_SCROLL_TO_EDITOR_SCRIPT, height=0)
6567 bar.caption(
6568 "How this dataset is read — where its files are, how its "
6569 "columns map onto the app's fields, the screen it was recorded on, and "
6570 "any metadata tables attached to it. The same questions the "
6571 "add-dataset screen asks, for a dataset that already exists."
6572 )
6575#: DATA-35 — set by a row action that genuinely changes what the app is showing
6576#: (opening a dataset, opening the editor), read at the top of the table
6577#: fragment. A widget callback may not call ``st.rerun``, and a *fragment* rerun
6578#: would redraw the table alone while the page under it still showed the old
6579#: dataset — so the callback asks and the fragment body does it.
6580_TABLE_NEEDS_APP_RERUN = "_dataset_table_needs_app_rerun"
6583#: UX-174 — the table's inspection seam: `dataset_table.row_record` per row, in
6584#: list order, rewritten every time the table draws.
6585DATASET_TABLE_ROWS_KEY = "_dataset_table_rows_current"
6586#: ``(column, descending)`` or absent (the list's own order). Plain state, not a
6587#: widget: it is written by the header buttons' callbacks.
6588_DATASET_TABLE_SORT_KEY = "_dataset_table_sort"
6589_DATASET_SEARCH_KEY = "dataset_table_search"
6590_DATASET_KIND_FILTER_KEY = "dataset_table_kinds"
6591_DATASET_LANGUAGE_FILTER_KEY = "dataset_table_languages"
6592#: Search and the Kind / Language filters appear only past this many rows — on
6593#: the short default list they would be controls with nothing to narrow. Ten,
6594#: since DATA-63 made OneStop four rows: the default list is eight.
6595_DATASET_FILTER_MIN_ROWS = 10
6596_DATASET_SORTABLE = ("Kind", "Dataset", *DATASET_COUNT_FIELDS, "Status")
6597#: Cell widths, in px, shared by the header and every row so the columns line
6598#: up. The name takes whatever is left (`width="stretch"`, with a CSS minimum).
6599_DATASET_KIND_W = 92
6600#: UX-178 — the name has a width of its own, so Status sits right after it and
6601#: the free space goes between Status and the counts (`_row_gap`).
6602_DATASET_NAME_W = 280
6603_DATASET_COUNT_W = 96
6604_DATASET_STATUS_W = 156 # BUG-113: fits the "Needs download" badge
6605_DATASET_ACTIONS_W = 40
6608def _dataset_row_slug(token: str) -> str:
6609 """A key-safe, stable id for one dataset's row widgets.
6611 Tokens are display names (spaces, punctuation, em dashes), and a widget key
6612 is also a CSS class, so the row's keys use a digest of the token instead —
6613 the same on every run and after any sort, which is what makes a click land
6614 on the dataset it was drawn for.
6615 """
6616 return hashlib.sha1(token.encode("utf-8")).hexdigest()[:12]
6619def _dataset_table_rows(
6620 *,
6621 active: str | None,
6622 words: pd.DataFrame | None,
6623 fixations: pd.DataFrame | None,
6624 raw_gaze: pd.DataFrame | None,
6625) -> list[DatasetRow]:
6626 """One `DatasetRow` per listed dataset, in the order they are offered."""
6627 entries = list(st.session_state.get("_data_source_entries") or [])
6628 kinds = dict(st.session_state.get("_data_source_kinds") or {})
6629 uploaded = set(st.session_state.get("_data_source_uploaded") or [])
6630 stored_uploads = dict(st.session_state.get("_datasets") or {})
6631 registry = public_dataset_registry()
6632 # DATA-32: a dataset that has left the list takes its remembered counts with
6633 # it — deleted, renamed, or a public corpus whose location was unset.
6634 forget_dataset_counts(keep={t for t in entries if t != UPLOAD_CHOICE})
6635 # The open corpus is not on disk and the demo is standing in for it: its
6636 # frames are the demo's, so they are not counted for it. An entry
6637 # remembered under its name *from these very frames* (counted that way
6638 # before this guard existed) is dropped; counts remembered from a load of
6639 # the real corpus have other fingerprints and are kept.
6640 placeholder = bool(st.session_state.get(_PLACEHOLDER_SHOWN_KEY))
6641 if placeholder and active:
6642 store = _counts_store()
6643 entry = store.get(active)
6644 stand_in = [frame_fingerprint(f) for f in (words, fixations, raw_gaze)]
6645 if isinstance(entry, dict) and entry.get("key") == stand_in:
6646 store.pop(active, None)
6647 rows: list[DatasetRow] = []
6648 for token in entries:
6649 if token in (UPLOAD_CHOICE, AUTHOR_CHOICE):
6650 continue # creation flows have their own buttons by the heading
6651 # DATA-32: counted once per version of a dataset and remembered, so a
6652 # row keeps its numbers without the frames being in memory. A corpus
6653 # nobody has opened yet is never loaded to fill its row.
6654 if token == active and placeholder:
6655 frames = (None, None, None)
6656 elif token == active:
6657 frames = (words, fixations, raw_gaze)
6658 elif token in stored_uploads:
6659 entry = stored_uploads.get(token) or {}
6660 frames = (
6661 entry.get("words"),
6662 entry.get("fixations"),
6663 entry.get("raw_gaze"),
6664 )
6665 elif token == MANUAL_SAMPLE_CHOICE:
6666 # Built from the session's draft, not read from disk, so it is as
6667 # cheap to count as to open — and nothing else would ever count it
6668 # on a server with no recovery cache. Its frames change with each
6669 # edit, so it publishes no figures and is counted here instead.
6670 frames = (*_manual_sample_frames(), None)
6671 else:
6672 frames = (None, None, None)
6673 about = dataset_about(token, registry)
6674 measured = remembered_dataset_counts(token, *frames)
6675 # DATA-36: a row that has never been opened shows the figures the corpus
6676 # publishes; the moment it is loaded, what loaded takes over.
6677 row_counts = dataset_row_counts(
6678 measured=measured, published=about.get("published_counts")
6679 )
6680 kind = _DATASET_KIND_LABELS.get(kinds.get(token, ""), "")
6681 if not kind and token in uploaded:
6682 kind = "Private"
6683 rows.append(
6684 DatasetRow(
6685 token=token,
6686 name=_dataset_display_name(token, registry),
6687 kind=kind,
6688 language=str(about.get("language") or ""),
6689 source=row_counts.source,
6690 counts=dict(row_counts.counts),
6691 exceeds_published=row_counts.exceeds_published,
6692 active=token == active,
6693 measured=bool(measured),
6694 status=_dataset_status(
6695 registry.get(token),
6696 stood_in_for=token == active and placeholder,
6697 ),
6698 # In memory: the open dataset, a stored upload, or one this
6699 # session already read (its loader's cache holds it).
6700 loaded=(token == active and not placeholder)
6701 or token in stored_uploads
6702 or token in st.session_state.get(LOADED_THIS_SESSION_KEY, ()),
6703 order=len(rows),
6704 )
6705 )
6706 return rows
6709def _dataset_status(spec: Mapping | None, *, stood_in_for: bool = False) -> str:
6710 """One row's **Status** — ``""`` (here) or what is missing (BUG-113).
6712 Asked the same way of every row, open or not: a corpus with files on disk
6713 has a ``files_present`` check in its registry entry — path stats only, never
6714 a read — and a missing set reads *Needs download* where the app can fetch
6715 it and *Needs setup* where it cannot. Bundled datasets and stored uploads
6716 have no check and are always here. The open row whose loader fell back to
6717 the demo (``stood_in_for``) is not here either, whatever the check says.
6718 """
6719 check = (spec or {}).get("files_present")
6720 if check is None:
6721 return dataset_table.NEEDS_SETUP if stood_in_for else ""
6722 try:
6723 present = bool(check())
6724 except _MANIFEST_ERRORS:
6725 present = False
6726 if present:
6727 return dataset_table.NEEDS_SETUP if stood_in_for else ""
6728 if (spec or {}).get("downloadable"):
6729 return dataset_table.NEEDS_DOWNLOAD
6730 return dataset_table.NEEDS_SETUP
6733#: The corpus whose row was clicked on a server that cannot fetch it, while
6734#: `_unreachable_dataset_dialog` explains why it did not open.
6735_UNREACHABLE_DATASET_KEY = "_dataset_unreachable_here"
6738def _unreachable_here(token: str) -> bool:
6739 """Whether ``token`` is a corpus this deployment can neither find nor fetch.
6741 The hosted demo (any server other machines can reach, `local_filesystem_enabled`)
6742 downloads nothing and takes no folder, so a corpus whose files are not
6743 already on it can never open there.
6744 """
6745 if local_filesystem_enabled():
6746 return False
6747 status = _dataset_status(public_dataset_registry().get(token))
6748 return status in (dataset_table.NEEDS_DOWNLOAD, dataset_table.NEEDS_SETUP)
6751def _dismiss_unreachable_dataset() -> None:
6752 st.session_state.pop(_UNREACHABLE_DATASET_KEY, None)
6755@st.dialog(
6756 f"{ICONS['desktop']} Open it on your own computer",
6757 on_dismiss=_dismiss_unreachable_dataset,
6758)
6759@guarded()
6760def _unreachable_dataset_dialog(token: str) -> None:
6761 """Why a public corpus did not open here, and where it does.
6763 Handled by the button's return value, as in `_delete_confirmation_dialog`
6764 (BUG-36): a dialog body is a fragment.
6765 """
6766 registry = public_dataset_registry()
6767 name = _dataset_display_name(token, registry)
6768 if (registry.get(token) or {}).get("downloadable"):
6769 st.markdown(
6770 f"**{name}** downloads the first time you open it, and this server "
6771 "doesn't download datasets. In the desktop app or a pip install it "
6772 "downloads once, with one click, and stays on your computer."
6773 )
6774 else:
6775 st.markdown(
6776 f"**{name}** is read from a folder on the computer running the app, "
6777 "and this server can't be pointed at one. Open it in the desktop "
6778 "app or a pip install instead."
6779 )
6780 st.markdown(
6781 f"{ICONS['desktop']} [Get the desktop app]({CITATION['desktop_url']}) ↗ "
6782 "(Windows, macOS, Linux), or install it with pip:"
6783 )
6784 st.code("pip install scanpath-studio\nscanpath-studio", language="bash")
6785 st.caption(
6786 "Running this server yourself? Place the files in its data location, "
6787 "or, on a trusted network, start it with `SCANPATH_LOCAL_FS=1`."
6788 )
6789 if st.button("OK", key="dataset_unreachable_ok", type="primary"):
6790 _dismiss_unreachable_dataset()
6791 st.rerun(scope="app")
6794def _open_dataset_row(token: str) -> None:
6795 """UX-78 — a click anywhere on a dataset's row opens it.
6797 UX-197: the table stays above ✏️ Edit dataset, and the editor edits the
6798 open dataset, so opening another one closes it — at once when nothing is
6799 unsaved, else through the editor's own *Leave without saving?*.
6800 """
6801 if token == st.session_state.get("data_source_choice"):
6802 return
6803 if _unreachable_here(token):
6804 # Opening it would only show the demo in its place, under a note
6805 # saying the files are missing. Say so before anything changes.
6806 st.session_state[_UNREACHABLE_DATASET_KEY] = token
6807 return
6808 st.session_state[_TABLE_NEEDS_APP_RERUN] = True
6809 if st.session_state.get(DATASET_EDITOR_OPEN_KEY):
6810 if _editor_is_dirty():
6811 st.session_state[_EDITOR_LEAVE_PENDING_KEY] = True
6812 st.session_state[_EDITOR_LEAVE_TARGET_KEY] = token
6813 return
6814 _close_dataset_editor()
6815 _select_dataset(token)
6818def _edit_open_dataset(token: str) -> None:
6819 """✏️ **Edit dataset**, under the open dataset's overview (UX-178).
6821 Raises ✏️ Edit dataset on it — the description, the column mapping, the
6822 recording setup, the source's options and location, the identity check and
6823 the metadata tables are all on that screen. `FOCUS_MAPPING_KEY` rides along
6824 for the mapping editor's "editing <name>". The manual sample has no mapping
6825 to edit: its editor is the authoring canvas.
6826 """
6827 if token == MANUAL_SAMPLE_CHOICE:
6828 _edit_manual_sample()
6829 return
6830 if not st.session_state.get(DATASET_EDITOR_OPEN_KEY):
6831 # UX-178 — the Name and Description fields are seeded on open; whatever
6832 # an editor left behind without Cancel or Save (a switch of dataset,
6833 # say) is not this edit's. An edit already open keeps its drafts.
6834 st.session_state.pop(EDITOR_NAME_FIELD_KEY, None)
6835 st.session_state.pop(EDITOR_PENDING_NAME_KEY, None)
6836 _drop_description_drafts()
6837 st.session_state[FOCUS_MAPPING_KEY] = token
6838 st.session_state[DATASET_EDITOR_OPEN_KEY] = True
6839 st.session_state[_EDITOR_SCROLL_KEY] = True
6842def _arm_dataset_row(pending_key: str, token: str) -> None:
6843 """Remove: arm the confirmation; the next run opens it.
6845 Remove in particular only *arms* (UX-54 r2, UX-79): an upload is not
6846 recoverable once dropped, so the confirmation does the work.
6847 """
6848 st.session_state[pending_key] = token
6851def _cycle_dataset_sort(column: str) -> None:
6852 current = st.session_state.get(_DATASET_TABLE_SORT_KEY)
6853 current = tuple(current) if isinstance(current, tuple | list) else None
6854 st.session_state[_DATASET_TABLE_SORT_KEY] = dataset_table.next_sort(current, column)
6857def _render_dataset_table_tools(box, rows: list[DatasetRow]) -> list[DatasetRow]:
6858 """Search and the Kind / Language filters, for a long list only.
6860 Returns the rows they leave. Nothing is drawn until the list is long enough
6861 to need narrowing, and each filter only when it has more than one value to
6862 choose between.
6863 """
6864 if len(rows) <= _DATASET_FILTER_MIN_ROWS:
6865 return rows
6866 tools = box.container(
6867 key="dataset_table_tools",
6868 horizontal=True,
6869 vertical_alignment="center",
6870 gap="small",
6871 )
6872 query = tools.text_input(
6873 "Search datasets",
6874 key=_DATASET_SEARCH_KEY,
6875 placeholder="Search by name",
6876 icon=ICONS["search"],
6877 label_visibility="collapsed",
6878 width=240,
6879 )
6880 kinds = sorted(
6881 {row.kind for row in rows if row.kind},
6882 key=dataset_table.KIND_ORDER.index,
6883 )
6884 picked_kinds = (
6885 tools.pills(
6886 "Kind",
6887 kinds,
6888 selection_mode="multi",
6889 key=_DATASET_KIND_FILTER_KEY,
6890 label_visibility="collapsed",
6891 )
6892 if len(kinds) > 1
6893 else []
6894 )
6895 languages = sorted({row.language for row in rows if row.language})
6896 picked_languages = (
6897 tools.multiselect(
6898 "Language",
6899 languages,
6900 key=_DATASET_LANGUAGE_FILTER_KEY,
6901 placeholder="Any language",
6902 label_visibility="collapsed",
6903 width=220,
6904 )
6905 if len(languages) > 1
6906 else []
6907 )
6908 return dataset_table.filter_rows(
6909 rows,
6910 query=query or "",
6911 kinds=picked_kinds or (),
6912 languages=picked_languages or (),
6913 )
6916def _render_dataset_table_head(grid, sort) -> None:
6917 """The header line — each sortable column's name is its sort button."""
6918 head = grid.container(
6919 key="dsrow_head",
6920 horizontal=True,
6921 vertical_alignment="center",
6922 gap="small",
6923 wrap=False,
6924 )
6926 def _sort_button(cell, column: str, help_text: str) -> None:
6927 icon = None
6928 if sort and sort[0] == column:
6929 icon = ICONS["sort_desc"] if sort[1] else ICONS["sort_asc"]
6930 cell.button(
6931 column,
6932 key=f"dataset_sort_{_field_slug(column)}",
6933 type="tertiary",
6934 icon=icon,
6935 icon_position="right",
6936 on_click=_cycle_dataset_sort,
6937 args=(column,),
6938 help=help_text,
6939 )
6941 _sort_button(
6942 head.container(key="dsh_kind", width=_DATASET_KIND_W),
6943 "Kind",
6944 "Sort by kind — Demo, Manual, Private, Public.",
6945 )
6946 _sort_button(
6947 head.container(key="dsh_name", width=_DATASET_NAME_W),
6948 "Dataset",
6949 "Sort by name. Click a row to open that dataset.",
6950 )
6951 _sort_button(
6952 head.container(key="dsh_status", width=_DATASET_STATUS_W),
6953 "Status",
6954 " ".join(
6955 f"**{label}** — {text}"
6956 for label, text in dataset_table.STATUS_EXPLANATIONS.items()
6957 ),
6958 )
6959 head.space("stretch")
6960 gaps = " ".join(
6961 f"**{label}** — {dataset_table.GAP_EXPLANATIONS[label]}"
6962 for label in (
6963 dataset_table.NOT_LOADED,
6964 dataset_table.NOT_REPORTED,
6965 dataset_table.NOT_APPLICABLE,
6966 dataset_table.UNKNOWN,
6967 )
6968 )
6969 for count_field in dataset_table.TABLE_COUNT_FIELDS:
6970 cell = head.container(
6971 key=f"dsh_{_field_slug(count_field)}",
6972 width=_DATASET_COUNT_W,
6973 horizontal=True,
6974 horizontal_alignment="right",
6975 )
6976 _sort_button(
6977 cell,
6978 count_field,
6979 f"Sort by {count_field.lower()}, largest first. Datasets without a "
6980 f"count sort last either way.\n\n{dataset_table.COUNTS_EXPLANATION}"
6981 f"\n\n{gaps}",
6982 )
6983 # The actions column has no title: its one button says what it does. The
6984 # cell is still drawn — an empty container is not — so the columns line up.
6985 head.container(key="dsh_actions", width=_DATASET_ACTIONS_W).markdown(
6986 '<span aria-hidden="true"> </span>', unsafe_allow_html=True
6987 )
6990def _field_slug(count_field: str) -> str:
6991 return count_field.lower().replace(" ", "_")
6994def _dataset_count_cell_html(row: DatasetRow, count_field: str) -> str:
6995 """One count cell: the grouped number, or its reason in a muted voice."""
6996 value = row.value(count_field)
6997 if value is not None:
6998 return f'<span class="sps-ds-num">{dataset_table.format_count(value)}</span>'
6999 return (
7000 f'<span class="sps-ds-num sps-ds-gap">{html.escape(row.gap(count_field))}'
7001 "</span>"
7002 )
7005def _render_dataset_table_row(grid, row: DatasetRow) -> None:
7006 """One dataset's line of the table.
7008 The whole line opens the dataset: its first child is a button stretched
7009 over the row by CSS (`styles.py`, *UX-174 r2*), and the cells drawn above it
7010 let a click through, except the one holding **Remove** — drawn for the
7011 datasets you added only.
7012 """
7013 slug = _dataset_row_slug(row.token)
7014 line = grid.container(
7015 # The open row's key carries `current`, which is what the tint keys on.
7016 key=f"dsrow_current_{slug}" if row.active else f"dsrow_{slug}",
7017 horizontal=True,
7018 vertical_alignment="center",
7019 gap="small",
7020 wrap=False,
7021 )
7022 line.button(
7023 f"Open {row.name}",
7024 key=f"dataset_open_{slug}",
7025 type="tertiary",
7026 on_click=_open_dataset_row,
7027 args=(row.token,),
7028 )
7029 kind = line.container(key=f"dsc_kind_{slug}", width=_DATASET_KIND_W)
7030 if row.kind:
7031 kind.markdown(f"{_DATASET_KIND_ICONS[row.kind]} {row.kind}")
7033 name = line.container(
7034 key=f"dsc_name_{slug}",
7035 width=_DATASET_NAME_W,
7036 horizontal=True,
7037 vertical_alignment="center",
7038 gap="xsmall",
7039 wrap=True,
7040 )
7041 name.markdown(
7042 f'<span class="sps-ds-name">{html.escape(row.name)}</span>',
7043 unsafe_allow_html=True,
7044 width="content",
7045 )
7046 if row.active:
7047 # A word, not only a colour: the badge says it, the tint repeats it.
7048 name.badge("Current", icon=ICONS["current"], color="blue")
7049 if row.exceeds_published:
7050 # DATA-36's discrepancy warning, kept — it is rare, and it is the one
7051 # thing on a row that says a number elsewhere is wrong.
7052 name.badge(
7053 "More than published",
7054 icon=ICONS["warning"],
7055 color="orange",
7056 help="This session loaded **more** than the published figure for "
7057 f"{', '.join(row.exceeds_published)}. A lab export can hold more "
7058 "than the public release; otherwise the figure is out of date.",
7059 )
7061 status = line.container(key=f"dsc_status_{slug}", width=_DATASET_STATUS_W)
7062 if row.status:
7063 # Something is missing before the dataset can open (BUG-113).
7064 status.badge(row.status, icon=ICONS["warning"], color="orange")
7065 else:
7066 status.markdown(
7067 f'<span class="sps-ds-status">{row.status_label}</span>',
7068 unsafe_allow_html=True,
7069 )
7071 line.space("stretch")
7072 for count_field in dataset_table.TABLE_COUNT_FIELDS:
7073 cell = line.container(
7074 key=f"dsc_{_field_slug(count_field)}_{slug}", width=_DATASET_COUNT_W
7075 )
7076 cell.markdown(
7077 _dataset_count_cell_html(row, count_field), unsafe_allow_html=True
7078 )
7080 actions = line.container(
7081 key=f"dsc_actions_{slug}",
7082 width=_DATASET_ACTIONS_W,
7083 horizontal=True,
7084 horizontal_alignment="right",
7085 vertical_alignment="center",
7086 )
7087 # Only a dataset you added can be removed. For the demo, a public corpus
7088 # or a local bundle, Remove only hid the row for the rest of the session —
7089 # nothing was deleted and nothing could bring it back — so they offer none.
7090 # The empty cell keeps the columns lined up — drawn with a space in it, as
7091 # the header's is, because an empty container is not drawn at all and the
7092 # row's numbers then sat right of an added dataset's (#374 F30).
7093 if row.token not in set(st.session_state.get("_data_source_uploaded") or []):
7094 actions.markdown(
7095 '<span aria-hidden="true"> </span>', unsafe_allow_html=True
7096 )
7097 return
7098 actions.button(
7099 f"Remove {row.name}",
7100 icon=ICONS["delete"],
7101 key=f"dataset_row_remove_{slug}",
7102 type="tertiary",
7103 on_click=_arm_dataset_row,
7104 args=(PENDING_DELETE_KEY, row.token),
7105 help=f"Delete {row.name}, its tables and annotations, after a "
7106 "confirmation. Your original files are not touched.",
7107 )
7110def dataset_table_scope_note(*, filtered: bool, stand_in_for: str | None) -> str | None:
7111 """The line under 📂 Available datasets when 📊 Stats counts something else
7112 (UX-203): a trial-filtered pool, or the bundled demo standing in for
7113 ``stand_in_for``, a corpus that isn't on disk. ``None`` when the row and
7114 Stats count the same thing.
7115 """
7116 if stand_in_for:
7117 return (
7118 f"Rows count whole datasets. {stand_in_for} isn't loaded, so Stats "
7119 "below counts the bundled demo shown in its place"
7120 + (", narrowed by the trial filters." if filtered else ".")
7121 )
7122 if filtered:
7123 return (
7124 "Whole datasets, before the trial filters — Stats below counts the "
7125 "filtered trials."
7126 )
7127 return None
7130@st.fragment
7131@guarded()
7132def render_dataset_table(
7133 host=None,
7134 *,
7135 active: str | None = None,
7136 words: pd.DataFrame | None = None,
7137 fixations: pd.DataFrame | None = None,
7138 raw_gaze: pd.DataFrame | None = None,
7139 scope_note: str | None = None,
7140) -> None:
7141 """📂 Available datasets — one focused row per dataset (UX-54 → UX-174).
7143 **Kind · Dataset · Status · Participants · Texts · Trials · Fixations ·
7144 Remove** (UX-178 moved Status beside the name it qualifies). A click anywhere on a row opens that dataset (UX-78); the open one
7145 carries a **Current** badge and a tint, and never moves. **Status** is
7146 whether the dataset can be opened now — *Loaded*, *Available*, *Needs download* or
7147 *Needs setup* — asked the same way of every row (BUG-113; see
7148 `_dataset_status`). Everything else about a dataset —
7149 Screens, Words and Gaze points, its description, renaming it, editing its
7150 setup — is in *What's in the dataset* under the table, for the open one.
7152 Built from widgets rather than an ``st.dataframe``: a grid cannot say *Not
7153 loaded* in a numeric column without sorting it as text. Sorting is ours
7154 instead (`dataset_table.sort_rows`, on the integers, missing values last),
7155 so it is numeric by construction; and every control is a real button keyed
7156 by its dataset, so a click after any sort acts on the row it was drawn in.
7158 **Counts are only shown for data already in memory, remembered, or
7159 published** — the open dataset (whose frames are passed in), every stored
7160 upload, anything counted earlier (DATA-32), and a corpus' published figures.
7161 Nothing is read from disk to fill a row.
7163 Args:
7164 host: Container to render into. Defaults to the fragment's own.
7165 active: The open dataset's entry token, marked in the table.
7166 words: The open dataset's word frame, for its counts.
7167 fixations: Its fixation frame.
7168 raw_gaze: Its raw-gaze frame.
7169 scope_note: A line under the rows saying the counts are whole
7170 datasets, given while 📊 Stats counts something else
7171 (`dataset_table_scope_note`, UX-203).
7172 """
7173 # DATA-35: Remove only opens a dialog, which costs a *fragment* rerun, not a
7174 # whole-app one. Opening a dataset asks for the app rerun here, because a
7175 # callback may not.
7176 if st.session_state.pop(_TABLE_NEEDS_APP_RERUN, False):
7177 st.rerun(scope="app")
7178 rows = _dataset_table_rows(
7179 active=active, words=words, fixations=fixations, raw_gaze=raw_gaze
7180 )
7181 # Inspection seam (tests, the debug panel): what each row says, by value.
7182 st.session_state[DATASET_TABLE_ROWS_KEY] = [
7183 dataset_table.row_record(row) for row in rows
7184 ]
7185 if not rows:
7186 return
7187 uploaded = set(st.session_state.get("_data_source_uploaded") or [])
7188 tokens = [row.token for row in rows]
7189 box = (host if host is not None else st).container(key="dataset_table")
7191 shown = _render_dataset_table_tools(box, rows)
7192 sort = st.session_state.get(_DATASET_TABLE_SORT_KEY)
7193 if not (
7194 isinstance(sort, tuple | list)
7195 and len(sort) == 2
7196 and sort[0] in _DATASET_SORTABLE
7197 ):
7198 sort = None
7199 grid = box.container(key="dataset_table_grid")
7200 _render_dataset_table_head(grid, sort)
7201 ordered = dataset_table.sort_rows(
7202 shown, sort[0] if sort else None, descending=bool(sort and sort[1])
7203 )
7204 for row in ordered:
7205 _render_dataset_table_row(grid, row)
7206 if not ordered:
7207 box.caption("No dataset matches the search and filters.")
7208 if scope_note:
7209 box.container(key="dataset_table_scope").caption(
7210 f"{ICONS['trial_filter']} {scope_note}"
7211 )
7213 _render_delete_confirmation(box, tokens, uploaded)
7214 if unreachable := st.session_state.get(_UNREACHABLE_DATASET_KEY):
7215 _unreachable_dataset_dialog(unreachable)
7216 if note := st.session_state.pop("_dataset_table_note", None):
7217 box.success(note)
7220def _rows_with_local_images(frame: pd.DataFrame) -> int:
7221 """How many rows of ``frame`` name a stimulus image that exists on disk.
7223 One ``os.path.isfile`` per **distinct** path rather than per row. The whole
7224 point of `data.resolve_stimulus_image_paths` probing once per placeholder
7225 tuple is lost if the caption it feeds then re-stats every row of a
7226 multi-million-row corpus.
7227 """
7228 paths = None if frame is None else frame.get("image_path")
7229 if paths is None or paths.empty:
7230 return 0
7231 counts = paths.dropna().astype(str).value_counts()
7232 return int(sum(rows for path, rows in counts.items() if os.path.isfile(path)))
7235def resolve_source_monitor(
7236 data_choice: str | None,
7237 words: pd.DataFrame | None,
7238 fixations: pd.DataFrame | None,
7239 *,
7240 own_setup: bool = True,
7241) -> tuple[int, int, bool]:
7242 """The presentation monitor for a data source: ``(width, height, authoritative)``.
7244 Lifted out of `seed_canvas_state` by **CMP-8 §1** so there is *one* source →
7245 monitor table rather than two. `authoritative` keeps its meaning: the source
7246 declares a real presentation monitor, so the canvas should snap to it rather
7247 than to data-derived extents (which undershoot, because text rarely fills the
7248 screen).
7250 A **stored upload** now answers for itself when it recorded a setup snapshot
7251 — before CMP-8 it declared nothing, so switching to one silently left the
7252 canvas on the *previous* source's monitor.
7254 ``words`` / ``fixations`` are read by the **last** branch alone, which
7255 estimates a canvas from data extents when nothing else declared one. Pass
7256 ``None`` for both to say "don't estimate": the answer degrades to
7257 `DEFAULT_FIGURE_SIZE`, still non-authoritative, and the row scan is skipped.
7258 Only a caller that will discard a non-authoritative size may do that.
7260 A built-in or public dataset whose recording setup the user saved on
7261 ✏️ Edit dataset answers with that screen, authoritatively — it is this
7262 dataset's setup now (`dataset_setup_override`). ``own_setup=False`` asks
7263 for what the source itself declares instead: the share link's elision
7264 (`url_state._link_defaults`) needs what a *recipient* resolves, and the
7265 recipient has no override.
7266 """
7267 if own_setup and (
7268 override := dataset_setup_override(setup_override_token(data_choice))
7269 ):
7270 return override.canvas_width, override.canvas_height, True
7271 # OneStop server bundle + bundled demo share the same experimental setup
7272 # (Dell U2715H, 2560x1440) — cited once in
7273 # `eyegenbench_geometry.DISPLAY_SPECS["onestop"]`.
7274 if data_choice in (ONESTOP_CHOICE, DEMO_CHOICE):
7275 return 2560, 1440, True
7276 if data_choice == MANUAL_SAMPLE_CHOICE:
7277 return (*_manual_sample_canvas(), True)
7278 if (monitor := _public_dataset_monitor(data_choice)) is not None:
7279 return monitor[0], monitor[1], True
7280 # A public corpus reached by its own registry *label* rather than through the
7281 # `Public datasets` picker still declares a monitor — `_public_dataset_monitor`
7282 # only answers for `PUBLIC_DATASETS_CHOICE`, but `data_source_choice` holds the
7283 # label (DATA-9's flat picker) and `compare_source` names B by label too.
7284 # `active_setup_snapshot` already compensates; without the same fallback here a
7285 # public corpus fell through to `compute_canvas_size` and reported its rounded
7286 # data extents as an ESTIMATED canvas. Cosmetic before CMP-11 (only the split
7287 # panels' `canvas_b` read it); load-bearing now that `setups_comparable` gates
7288 # the overlay on the canvas, since it refused the very pairs CMP-11 exists to
7289 # allow — and it also made A's resolution scan the whole corpus per rerun.
7290 if declared := (public_dataset_registry().get(data_choice) or {}).get("monitor"):
7291 return int(declared[0]), int(declared[1]), True
7292 if data_choice == MULTIPLEYE_BUNDLE_CHOICE:
7293 # MultiplEYE server bundle = the same native MultiplEYE export as the
7294 # public source; coordinates are offset onto the centered stimulus on
7295 # the real 1920x1080 monitor, so snap the canvas to it (true-to-scale),
7296 # exactly like the public MultiplEYE registry entry's monitor.
7297 from scanpath_studio.datasets import MULTIPLEYE_MONITOR
7299 return MULTIPLEYE_MONITOR[0], MULTIPLEYE_MONITOR[1], True
7300 stored = (st.session_state.get("_datasets") or {}).get(data_choice)
7301 if isinstance(stored, dict) and isinstance(stored.get("setup"), dict):
7302 snapshot = SetupSnapshot.from_dict(stored["setup"], fallback=SetupSnapshot())
7303 # Authoritative only when the screen was actually *known*: an assumed or
7304 # estimated canvas must not snap over a canvas the user has since tuned.
7305 return (
7306 snapshot.canvas_width,
7307 snapshot.canvas_height,
7308 snapshot.screen_provenance is Provenance.MEASURED,
7309 )
7310 if data_choice is None or data_choice == UPLOAD_CHOICE:
7311 # Uploaded data (the setup wizard passes data_choice=None) defaults to a
7312 # common 1440p monitor until the Recording-setup step says otherwise.
7313 return DEFAULT_FIGURE_SIZE[0], DEFAULT_FIGURE_SIZE[1], False
7314 derived_w, derived_h = compute_canvas_size(words, fixations)
7315 return derived_w, derived_h, False
7318def capture_setup_snapshot(
7319 provenance: Mapping[str, Provenance] | None = None,
7320) -> SetupSnapshot:
7321 """The resolved ``global_*`` geometry as a `SetupSnapshot` (CMP-8 §1).
7323 Reads the same keys `seed_canvas_state` resolves, so there is no second list
7324 of key names to keep in sync. ``provenance`` overrides the per-group
7325 provenance — the wizard passes what the user answered; a built-in corpus that
7326 declares its own monitor passes ``MEASURED``.
7327 """
7328 ss = st.session_state
7329 base = SetupSnapshot()
7330 groups = dict(base.provenance)
7331 groups.update(provenance or {})
7332 return SetupSnapshot(
7333 canvas_width=int(ss.get("global_canvas_width", base.canvas_width)),
7334 canvas_height=int(ss.get("global_canvas_height", base.canvas_height)),
7335 monitor_width_mm=float(
7336 ss.get("global_monitor_width_mm", base.monitor_width_mm)
7337 ),
7338 viewing_distance_mm=float(
7339 ss.get("global_viewing_distance_mm", base.viewing_distance_mm)
7340 ),
7341 base_font_size=int(ss.get("global_base_font_size", base.base_font_size)),
7342 font_family=str(ss.get("global_font_family", base.font_family)),
7343 line_spacing=float(ss.get("global_line_spacing", base.line_spacing)),
7344 scale_text_to_boxes=bool(
7345 ss.get("global_scale_text_to_boxes", base.scale_text_to_boxes)
7346 ),
7347 screen_provenance=groups["screen"],
7348 geometry_provenance=groups["geometry"],
7349 text_provenance=groups["text"],
7350 )
7353def setup_override_token(data_choice: str | None) -> str | None:
7354 """The dataset a recording-setup override is filed under: the public
7355 corpus' label behind ``PUBLIC_DATASETS_CHOICE``, else the choice itself."""
7356 if data_choice == PUBLIC_DATASETS_CHOICE:
7357 return st.session_state.get("public_dataset_choice") or None
7358 return data_choice
7361def dataset_setup_override(token: str | None) -> SetupSnapshot | None:
7362 """The recording setup the user saved for a built-in or public dataset, or
7363 ``None`` when they saved none (the corpus' own declaration stands).
7365 An upload's setup lives on its own ``_datasets`` entry, so a name that is an
7366 upload never answers here, even if an override was once filed under it.
7367 """
7368 if not token or token in (st.session_state.get("_datasets") or {}):
7369 return None
7370 payload = (st.session_state.get(DATASET_SETUP_OVERRIDES_KEY) or {}).get(token)
7371 if not isinstance(payload, dict):
7372 return None
7373 return SetupSnapshot.from_dict(payload, fallback=SetupSnapshot())
7376def setup_override_session_values(snapshot: SetupSnapshot) -> dict:
7377 """The ``global_*`` values a saved setup puts on the figure — the keys the
7378 add flow publishes, plus the DPI those imply."""
7379 return {
7380 "global_canvas_width": int(snapshot.canvas_width),
7381 "global_canvas_height": int(snapshot.canvas_height),
7382 "global_monitor_width_mm": float(snapshot.monitor_width_mm),
7383 "global_viewing_distance_mm": float(snapshot.viewing_distance_mm),
7384 "global_display_dpi": round(
7385 float(snapshot.canvas_width) / (float(snapshot.monitor_width_mm) / 25.4),
7386 2,
7387 ),
7388 "global_base_font_size": int(snapshot.base_font_size),
7389 "global_font_family": str(snapshot.font_family),
7390 "global_line_spacing": float(snapshot.line_spacing),
7391 "global_scale_text_to_boxes": bool(snapshot.scale_text_to_boxes),
7392 }
7395def _restore_setup_override_stash() -> None:
7396 """Put back what the ``global_*`` keys held before an override was applied."""
7397 st.session_state.pop(SETUP_OVERRIDE_FOR_KEY, None)
7398 stashed = st.session_state.pop(SETUP_OVERRIDE_RESTORE_KEY, None)
7399 for key, value in (stashed or {}).items():
7400 if value is None:
7401 st.session_state.pop(key, None)
7402 else:
7403 st.session_state[key] = value
7406def _apply_setup_override(
7407 token: str, snapshot: SetupSnapshot, skip: frozenset = frozenset()
7408) -> None:
7409 """Write ``snapshot`` onto the figure as ``token``'s setup, remembering what
7410 it replaced. ``skip`` — keys a share link just seeded, which win."""
7411 if st.session_state.get(SETUP_OVERRIDE_FOR_KEY) != token:
7412 # The first override of a run of them keeps the pre-override state.
7413 st.session_state.setdefault(
7414 SETUP_OVERRIDE_RESTORE_KEY,
7415 {
7416 key: None if key in skip else st.session_state.get(key)
7417 for key in SETUP_OVERRIDE_SESSION_KEYS
7418 },
7419 )
7420 for key, value in setup_override_session_values(snapshot).items():
7421 if key not in skip:
7422 st.session_state[key] = value
7423 st.session_state[SETUP_OVERRIDE_FOR_KEY] = token
7426def save_dataset_setup_override(token: str, payload: dict | None) -> None:
7427 """✅ Save changes for a built-in or public dataset's Recording setup.
7429 ``payload`` (a ``SetupSnapshot.to_dict()``) becomes the dataset's own setup
7430 and applies to the figure at once, as an upload's saved setup does;
7431 ``None`` drops it — *Reset to source setup* — and puts back what the figure
7432 showed before it, so the corpus' declared screen snaps in again. Only this
7433 dataset's entry changes: another dataset's override is never touched.
7434 """
7435 overrides = dict(st.session_state.get(DATASET_SETUP_OVERRIDES_KEY) or {})
7436 if payload is None:
7437 overrides.pop(token, None)
7438 st.session_state[DATASET_SETUP_OVERRIDES_KEY] = overrides
7439 if st.session_state.get(SETUP_OVERRIDE_FOR_KEY) == token:
7440 _restore_setup_override_stash()
7441 # Re-snap the canvas to what the source itself declares.
7442 st.session_state.pop("_canvas_seeded_for", None)
7443 return
7444 snapshot = SetupSnapshot.from_dict(payload, fallback=SetupSnapshot())
7445 overrides[token] = snapshot.to_dict()
7446 st.session_state[DATASET_SETUP_OVERRIDES_KEY] = overrides
7447 _apply_setup_override(token, snapshot)
7450def _declared_setup_snapshot(choice: str | None) -> SetupSnapshot | None:
7451 """What a built-in source itself declares (``active_setup_snapshot``'s
7452 answer before overrides existed), or ``None`` when it declares nothing."""
7453 declared = (
7454 choice in (ONESTOP_CHOICE, DEMO_CHOICE, MULTIPLEYE_BUNDLE_CHOICE)
7455 or _public_dataset_monitor(choice) is not None
7456 # A public corpus reached by its own label (the DATA-3 OneStop source)
7457 # rather than through the `Public datasets` picker still declares a
7458 # monitor in the registry.
7459 or bool((public_dataset_registry().get(choice) or {}).get("monitor"))
7460 )
7461 if not declared:
7462 return None
7463 snapshot = capture_setup_snapshot(
7464 {
7465 # The corpus declares its presentation monitor, so the screen is
7466 # measured. The physical size / viewing distance are *not* — no
7467 # registry entry records them, so they stay honestly "assumed".
7468 "screen": Provenance.MEASURED,
7469 "geometry": Provenance.ASSUMED,
7470 "text": Provenance.MEASURED,
7471 }
7472 )
7473 token = setup_override_token(choice)
7474 if token and st.session_state.get(SETUP_OVERRIDE_FOR_KEY) == token:
7475 # The live keys hold this dataset's override: the source's own values
7476 # are the ones it replaced, and its screen the one it declares.
7477 stashed = st.session_state.get(SETUP_OVERRIDE_RESTORE_KEY) or {}
7478 width, height, _ = resolve_source_monitor(choice, None, None, own_setup=False)
7479 fields = {"canvas_width": int(width), "canvas_height": int(height)}
7480 for key, name, cast in (
7481 ("global_monitor_width_mm", "monitor_width_mm", float),
7482 ("global_viewing_distance_mm", "viewing_distance_mm", float),
7483 ("global_base_font_size", "base_font_size", int),
7484 ("global_font_family", "font_family", str),
7485 ("global_line_spacing", "line_spacing", float),
7486 ("global_scale_text_to_boxes", "scale_text_to_boxes", bool),
7487 ):
7488 value = stashed.get(key)
7489 fields[name] = (
7490 cast(value) if value is not None else getattr(SetupSnapshot(), name)
7491 )
7492 snapshot = replace(snapshot, **fields)
7493 return snapshot
7496@st.cache_data(show_spinner=False, max_entries=8)
7497def _c_canvas_size(_words, _fixations, fingerprints: tuple) -> tuple[int, int]:
7498 """`compute_canvas_size`, cached on the frames' fingerprints."""
7499 return compute_canvas_size(_words, _fixations)
7502def cached_canvas_size(
7503 words: pd.DataFrame | None, fixations: pd.DataFrame | None
7504) -> tuple[int, int]:
7505 """The data-extent screen estimate, without rescanning every row per rerun
7506 — ✏️ Edit dataset asks it on every render while its form is open."""
7507 return _c_canvas_size(
7508 words, fixations, (frame_fingerprint(words), frame_fingerprint(fixations))
7509 )
7512def source_setup_snapshot(
7513 data_choice: str | None,
7514 words: pd.DataFrame | None = None,
7515 fixations: pd.DataFrame | None = None,
7516) -> SetupSnapshot:
7517 """A built-in or public dataset's setup **as its source states it** — what
7518 ✏️ Edit dataset's *Reset to source setup* returns to.
7520 A corpus that declares its monitor reports it as measured; one that does
7521 not (a prepared benchmark corpus whose manifest invents its screen) gets
7522 the extent of its data, estimated, as `compare_source.snapshot_for` does.
7523 """
7524 declared = _declared_setup_snapshot(data_choice)
7525 if declared is not None:
7526 return declared
7527 width, height, authoritative = resolve_source_monitor(
7528 data_choice, None, None, own_setup=False
7529 )
7530 if not authoritative and (words is not None or fixations is not None):
7531 width, height = cached_canvas_size(words, fixations)
7532 return SetupSnapshot(
7533 canvas_width=int(width),
7534 canvas_height=int(height),
7535 screen_provenance=(
7536 Provenance.MEASURED if authoritative else Provenance.ESTIMATED
7537 ),
7538 geometry_provenance=Provenance.ASSUMED,
7539 text_provenance=Provenance.ASSUMED,
7540 )
7543def active_setup_snapshot(
7544 data_choice: str | None = None,
7545) -> SetupSnapshot | None:
7546 """A source's recorded setup, or ``None`` when it has none.
7548 A stored upload carries the snapshot the wizard captured; a built-in or
7549 public dataset whose setup the user saved on ✏️ Edit dataset reports that
7550 (`dataset_setup_override`); a built-in corpus that declares a monitor
7551 reports it as ``MEASURED``; anything else has nothing to say, and saying
7552 nothing is the honest answer (a caller must not print "assumed 2560x1440"
7553 for a corpus that never claimed one).
7555 ``data_choice`` defaults to the active source, but callers that already know
7556 which source they are describing — `_build_share_query` is handed one —
7557 should pass it rather than re-reading session state.
7558 """
7559 choice = (
7560 data_choice
7561 if data_choice is not None
7562 else st.session_state.get("data_source_choice")
7563 )
7564 stored = (st.session_state.get("_datasets") or {}).get(choice)
7565 if isinstance(stored, dict) and isinstance(stored.get("setup"), dict):
7566 return SetupSnapshot.from_dict(stored["setup"], fallback=SetupSnapshot())
7567 if (override := dataset_setup_override(setup_override_token(choice))) is not None:
7568 return override
7569 if (declared := _declared_setup_snapshot(choice)) is not None:
7570 return declared
7571 payload = st.session_state.get("_wizard_setup_snapshot")
7572 if isinstance(payload, dict):
7573 return SetupSnapshot.from_dict(payload, fallback=SetupSnapshot())
7574 return None
7577def seed_canvas_state(
7578 words_filtered: pd.DataFrame,
7579 fixations_filtered: pd.DataFrame,
7580 data_choice: str | None = None,
7581) -> tuple[int, int, int, str, float, bool]:
7582 """Resolve the canvas / typography settings without rendering any widget.
7584 Split out of `render_canvas_controls` by **VIZ-31**, which moved that
7585 panel from the sidebar into the Scanpath rail. The rail renders inside
7586 `tabs.render_single_trial_tab`, i.e. *after* `main` has to know the canvas
7587 size and font in order to build `viz_settings` and dispatch a view — and the
7588 Corpus view has no rail at all. This function does every session-state write
7589 the panel used to do on its way past (source-driven monitor/font snapping and
7590 the default seeding), then reads the resolved values back out, so both
7591 callers agree and neither has to render to learn them.
7593 Seeding is `controls._pin` (a plain default-if-absent); what keeps these
7594 keys alive on a view that never renders their widgets is the widgets' own
7595 `persist_state="session"` — see the comment on that block.
7597 Returns:
7598 Tuple of (canvas_width, canvas_height, base_font_size, font_family,
7599 line_spacing, scale_text_to_boxes). The text-sizing pair keeps the reading
7600 text true-to-scale: see `plots._word_label_font_px`.
7601 """
7602 # OneStop server bundle + bundled demo share the same experimental setup
7603 # (Dell U2715H, 2560x1440 — cited in
7604 # `eyegenbench_geometry.DISPLAY_SPECS["onestop"]`). Data-derived extents
7605 # undershoot — text only fills part of the screen — so hard-default to the
7606 # real monitor here.
7607 # ``monitor_is_authoritative`` = the source declares a real presentation
7608 # monitor (OneStop/demo or a public-dataset registry entry), so the canvas
7609 # should snap to it rather than to data-derived extents.
7610 #
7611 # The frames are withheld once the canvas keys exist, because that is the
7612 # only thing they are read for. `resolve_source_monitor`'s last-resort
7613 # branch scans every row of both to *estimate* a canvas from data extents —
7614 # tens of milliseconds per rerun on a full corpus — and a source that needs
7615 # estimating is by definition not authoritative, so the number it returns
7616 # reaches only `defaults` below, where `_pin` is a setdefault and `_resolved`
7617 # prefers the stored key. First run derives it; every rerun after that was
7618 # paying for a value it threw away. `tabs._compare_setups` withheld the same
7619 # scan from B's side for the same reason (CMP-11).
7620 canvas_seeded = {"global_canvas_width", "global_canvas_height"} <= set(
7621 st.session_state
7622 )
7623 # The source's own screen: a recording setup the user saved for it is
7624 # applied after these snaps (below), so what it replaces — and puts back on
7625 # leaving — is the source's canvas, not its own.
7626 default_canvas_w, default_canvas_h, monitor_is_authoritative = (
7627 resolve_source_monitor(
7628 data_choice,
7629 None if canvas_seeded else words_filtered,
7630 None if canvas_seeded else fixations_filtered,
7631 own_setup=False,
7632 )
7633 )
7634 canvas_width = min(max(default_canvas_w, 100), 10000)
7635 canvas_height = min(max(default_canvas_h, 100), 10000)
7636 # Seed the data-derived defaults so the inputs render without a `value=`
7637 # argument — that keeps the keys assignable by the plot-config restore
7638 # (app._restore_plot_config) without Streamlit's "default value but also set
7639 # via Session State API" warning.
7640 #
7641 # For a source with an authoritative monitor, snap the canvas to it whenever
7642 # that source *changes* (selecting a public dataset, switching PoTeC↔MultiplEYE,
7643 # or a public dataset whose registered monitor was updated): a plain
7644 # ``setdefault`` would let a previously-seeded canvas stick, so a returning
7645 # session would keep the old monitor and render the corpus off-scale. Manual
7646 # canvas edits and plot-config restores within the same source are preserved
7647 # (the key is unchanged, so the snap doesn't re-fire).
7648 #
7649 # DATA-27 (Task 11R): `public_dataset_choice` is a correct per-corpus key on
7650 # its own again, now that every prepared benchmark corpus is its own registry
7651 # entry — the extra `eyegenbench_dataset` component R30 needed (when one
7652 # source fronted many corpora and this key never changed between them) is
7653 # gone with the source that made it necessary.
7654 #
7655 # EXP-19 — except where a share link has just said otherwise. A link seeds
7656 # these keys before this first run gets here, and carries the canvas / font
7657 # only when the sender's differs from the corpus', so snapping over them
7658 # would undo exactly the part of the link that was worth sending. The link
7659 # names what it seeded *and for which source*; the first seeding consumes
7660 # that, and honours it only for the linked source — a link that fell back to
7661 # another corpus, or a first run that never got this far, must not leave the
7662 # next source opened without its own monitor.
7663 source_key = (data_choice, st.session_state.get("public_dataset_choice"))
7664 from_link = link_setup_keys_for(source_key)
7665 # A recording setup the user saved for a built-in or public dataset is that
7666 # dataset's alone: leaving it puts back what the figure held before — first,
7667 # so the snaps below then answer for the next source as they always have.
7668 override_token = setup_override_token(data_choice)
7669 applied_for = st.session_state.get(SETUP_OVERRIDE_FOR_KEY)
7670 if applied_for is not None and applied_for != override_token:
7671 _restore_setup_override_stash()
7672 if monitor_is_authoritative and st.session_state.get("_canvas_seeded_for") != (
7673 source_key
7674 ):
7675 for key, value in (
7676 ("global_canvas_width", canvas_width),
7677 ("global_canvas_height", canvas_height),
7678 ):
7679 if key not in from_link:
7680 st.session_state[key] = value
7681 st.session_state["_canvas_seeded_for"] = source_key
7682 # The canvas pair itself is pinned with the rest of the defaults below.
7684 # Authoritative reading font: MultiplEYE stamps the stimulus FONT_SIZE + family
7685 # from its config onto the words. Snap the font controls to it when the source
7686 # changes (same gate as the canvas), so the reading text renders at the exact
7687 # size and (CJK) typeface the stimulus images were drawn with. We also turn off
7688 # "scale text to boxes" since the precise px is known — box geometry can only
7689 # approximate it. Manual edits within a source stick (the key is unchanged, so
7690 # the snap doesn't re-fire); a returning source re-snaps to the known font.
7691 #
7692 # BUG-50 — **and the snap is undone on the way out.** It used to be gated on
7693 # ``font_px is not None``, so a source that declares no typeface skipped the
7694 # whole block and simply inherited the previous corpus'. Demo → MultiplEYE →
7695 # Demo therefore came back with MultiplEYE's px, its CJK stack, and
7696 # *scale-to-boxes off* — the demo's reading text visibly smaller than it had
7697 # been, with nothing on screen saying why, and it stayed that way for the
7698 # rest of the session (and into the recovery cache, and into any link or
7699 # config saved from it).
7700 #
7701 # The general shape, worth stating because the other two source-switch snaps
7702 # in this file share it: a snap writes **session**-scoped state to express a
7703 # **source**-scoped fact. That only stays consistent if every source answers
7704 # the question. The canvas one does (`resolve_source_monitor` always returns
7705 # a canvas) and so is self-correcting by luck rather than by design; this one
7706 # cannot, because most corpora ship no typeface — so it has to remember what
7707 # it overwrote and put it back.
7708 #
7709 # Restoring the *stashed* values rather than the factory defaults is what
7710 # keeps a hand-tuned font across a detour: tune the demo, look at MultiplEYE,
7711 # come back, and your own size is still there. Stashed only on the **first**
7712 # snap of a run of them, so MultiplEYE → another font-declaring corpus →
7713 # Demo restores the pre-MultiplEYE state and not MultiplEYE's.
7714 if st.session_state.get("_font_seeded_for") != source_key:
7715 # Read only when the snap can fire: on a font-declaring corpus it is a
7716 # numeric parse over every word row, and every other run discards it.
7717 font_px, font_css = _dataset_font(words_filtered)
7718 if font_px is not None:
7719 # A value a link seeded is this source's own, not something to put
7720 # back on the way out — so it is stashed as absent, and leaving the
7721 # corpus restores the factory value rather than the corpus' font.
7722 st.session_state.setdefault(
7723 _FONT_SNAP_RESTORE_KEY,
7724 {
7725 key: None if key in from_link else st.session_state.get(key)
7726 for key in _FONT_SNAP_KEYS
7727 },
7728 )
7729 snapped = {
7730 "global_base_font_size": int(min(max(round(font_px), 6), 72)),
7731 "global_scale_text_to_boxes": False,
7732 }
7733 if font_css:
7734 snapped["global_font_family"] = font_css
7735 for key, value in snapped.items():
7736 if key not in from_link:
7737 st.session_state[key] = value
7738 elif (
7739 stashed := st.session_state.pop(_FONT_SNAP_RESTORE_KEY, None)
7740 ) is not None:
7741 for key, value in stashed.items():
7742 if value is None:
7743 # Absent before the snap — drop it so the `defaults` pin a
7744 # few lines down restores the factory value this same run,
7745 # rather than freezing whatever the last corpus left.
7746 st.session_state.pop(key, None)
7747 else:
7748 st.session_state[key] = value
7749 # Not in the `font_px is None` + never-snapped case: a fresh session, a
7750 # deep link or a restored config may have set these keys deliberately,
7751 # and nothing has overwritten them, so there is nothing to undo.
7752 st.session_state["_font_seeded_for"] = source_key
7754 # …and entering a dataset with a saved setup applies it, after the canvas
7755 # and font snaps so it is what the figure shows. Once per visit: a value
7756 # tuned on the rail afterwards stays until the dataset is left.
7757 if (
7758 override_token
7759 and st.session_state.get(SETUP_OVERRIDE_FOR_KEY) != override_token
7760 and (override := dataset_setup_override(override_token)) is not None
7761 ):
7762 _apply_setup_override(override_token, override, frozenset(from_link))
7764 # The remaining widget defaults. Each of these used to be `setdefault`ed
7765 # inline, immediately above its own widget; seeding them here is what lets a
7766 # caller resolve the settings without rendering the panel.
7767 #
7768 # Seeding alone is NOT what keeps these alive. Since VIZ-31 these widgets
7769 # render only in the Scanpath rail, and Streamlit drops a widget's key from
7770 # session state at the end of any run in which the widget did not render — so
7771 # one trip through **Corpus Analysis** (no rail) would prune all fourteen and
7772 # the next run would seed the factory default over the user's canvas, font,
7773 # text colour and background, permanently. What prevents that is
7774 # `persist_state="session"` on each of those widgets (ENG-36; it replaced a
7775 # hand-rolled re-assert-every-run workaround). Six of the fourteen are
7776 # share-link / saved-config wire format, so this is not cosmetic. Pinned by
7777 # `test_canvas_settings_survive_a_corpus_analysis_round_trip`.
7778 ss = st.session_state
7779 bg_options = list(BACKGROUND_PRESETS.keys()) + ["Custom…"]
7780 if ss.get("global_bg_choice") not in bg_options:
7781 ss.pop("global_bg_choice", None)
7782 # One table drives both the pin and the read-back. `_pin` swallows the
7783 # StreamlitAPIException raised when a key's widget was already built earlier
7784 # in the run, so a pinned key is *not* guaranteed to land — every read below
7785 # therefore goes through `_resolved`, never `ss[...]`, or a swallowed write
7786 # would surface as a KeyError that takes the whole app down.
7787 defaults = {
7788 "global_canvas_width": canvas_width,
7789 "global_canvas_height": canvas_height,
7790 **SETUP_DEFAULTS,
7791 "global_scale_text_to_boxes": True,
7792 "global_line_spacing": float(DEFAULT_LINE_SPACING),
7793 "global_font_family": FONT_FAMILY,
7794 "global_text_color": WORD_LABEL_COLOR,
7795 "global_bg_choice": bg_options[0],
7796 # Pinned here as well as in the render path: its picker exists only while
7797 # the choice is "Custom…", so it is the one key with no other keeper —
7798 # without this a custom background is lost the first time the user opens
7799 # Corpus Analysis and the choice silently falls back to a preset.
7800 "global_bg_custom": DEFAULT_BACKGROUND_COLOR,
7801 }
7803 def _resolved(key):
7804 return ss.get(key, defaults[key])
7806 for key, default in defaults.items():
7807 _pin(key, default)
7808 # Derived from the two above it, so it is pinned after them.
7809 _pin(
7810 "global_display_dpi",
7811 round(
7812 float(_resolved("global_canvas_width"))
7813 / (float(_resolved("global_monitor_width_mm")) / 25.4),
7814 2,
7815 ),
7816 )
7817 # Point-specified stimulus typography converts to px through the DPI above.
7818 # The rendering path recomputes this from its own widget values (which is
7819 # what makes an edit apply the same run); this keeps the non-rendering
7820 # callers on the same number.
7821 if not _resolved("global_scale_text_to_boxes") and _resolved(
7822 "global_use_stimulus_font_pt"
7823 ):
7824 ss["global_base_font_size"] = int(
7825 min(
7826 max(
7827 round(
7828 font_pt_to_px(
7829 float(_resolved("global_stimulus_font_pt")),
7830 float(ss.get("global_display_dpi", 96.0)),
7831 )
7832 ),
7833 6,
7834 ),
7835 72,
7836 )
7837 )
7838 return (
7839 int(_resolved("global_canvas_width")),
7840 int(_resolved("global_canvas_height")),
7841 int(_resolved("global_base_font_size")),
7842 str(_resolved("global_font_family")),
7843 float(_resolved("global_line_spacing")),
7844 bool(_resolved("global_scale_text_to_boxes")),
7845 )
7848#: The CSS stack UX-163's *Multilingual* button writes into the text font — a
7849#: CJK / Hebrew / Arabic-capable fallback (PRE-6).
7850_MULTILINGUAL_FONT_STACK = (
7851 "'Noto Sans', 'Noto Sans Hebrew', 'Noto Sans Arabic', "
7852 "'Noto Sans CJK SC', 'Arial Unicode MS', sans-serif"
7853)
7856def _rail_text_rows(
7857 host,
7858 *,
7859 seeded: tuple,
7860 display_dpi: float,
7861 words_filtered: pd.DataFrame,
7862 font_css,
7863 disabled: bool,
7864 section: str | None,
7865) -> tuple[int, str, float, bool]:
7866 """The rail's typography rows, under 📄 Stimulus → *Text* (UX-163).
7868 Four captioned rows (`_sub_row`) instead of up to nine that came and went:
7870 * *Fit* — **Scale to boxes** and the line spacing it divides a box by
7871 (greyed while it is off);
7872 * *Size* — the unit (px / pt) and the size. While the text is fitted to the
7873 boxes the size is the axis, legend and fallback text's, in px, so the unit
7874 greys; otherwise it is the reading text's, and a size in points is
7875 converted with the dataset DPI (px = pt × DPI ÷ 72);
7876 * *Font* — the font family and the *Multilingual* stack;
7877 * *Color* — the text colour, then the plot background (and its custom
7878 colour, greyed unless *Custom…* is picked).
7880 ``disabled`` greys every row while the Text layer is off (UX-97's contract:
7881 the settings stay readable, and their stored values are untouched).
7882 Returns ``(base_font_size, font_family, line_spacing, scale_text_to_boxes)``.
7883 """
7884 off_reason = (
7885 f"{ICONS['warning']} **Text** is off — turn it on to change this. Your "
7886 "settings are kept either way."
7887 if disabled
7888 else ""
7889 )
7891 def tip(text: str) -> str:
7892 return f"{off_reason}\n\n{text}" if off_reason else text
7894 with host:
7895 fit = _sub_row(
7896 "Fit",
7897 section=section,
7898 section_help="How the reading text is drawn.",
7899 caption_help=tip(
7900 "**Scale to boxes**: size the text from the word-box height (box "
7901 "height ÷ line spacing). Spacing: how many text lines one box "
7902 "spans (OneStop: 3). Untick to set a fixed size below."
7903 ),
7904 )
7905 fit_col, spacing_cap_col, spacing_col = fit.columns(
7906 [0.55, 0.2, 0.25], gap=_LABEL_GAP, vertical_alignment="center"
7907 )
7908 scale_text_to_boxes = fit_col.checkbox(
7909 "Scale to boxes",
7910 key="global_scale_text_to_boxes",
7911 persist_state="session",
7912 disabled=disabled,
7913 )
7914 _sub_caption(spacing_cap_col, "Spacing")
7915 line_spacing = spacing_col.number_input(
7916 "Line spacing",
7917 min_value=1.0,
7918 max_value=10.0,
7919 step=0.5,
7920 key="global_line_spacing",
7921 persist_state="session",
7922 disabled=disabled or not scale_text_to_boxes,
7923 label_visibility="collapsed",
7924 )
7926 size = _sub_row(
7927 "Size",
7928 caption_help=tip(
7929 "With **Scale to boxes** on: the size of the axis and legend text. "
7930 "Off: the reading text's size, in px or pt (px = pt × DPI ÷ 72)."
7931 ),
7932 )
7933 unit_col, size_col = size.columns(
7934 [0.5, 0.5], gap=_LABEL_GAP, vertical_alignment="center"
7935 )
7936 use_pt = unit_col.segmented_control(
7937 "Font unit",
7938 options=[False, True],
7939 format_func=lambda use_pt: "pt" if use_pt else "px",
7940 key="global_use_stimulus_font_pt",
7941 persist_state="session",
7942 disabled=disabled or scale_text_to_boxes,
7943 label_visibility="collapsed",
7944 )
7945 if not scale_text_to_boxes and use_pt:
7946 stimulus_font_pt = size_col.number_input(
7947 "Font size (pt)",
7948 min_value=4.0,
7949 max_value=144.0,
7950 step=0.5,
7951 key="global_stimulus_font_pt",
7952 persist_state="session",
7953 disabled=disabled,
7954 label_visibility="collapsed",
7955 )
7956 st.session_state["global_base_font_size"] = int(
7957 min(max(round(font_pt_to_px(stimulus_font_pt, display_dpi)), 6), 72)
7958 )
7959 base_font_size = int(st.session_state["global_base_font_size"])
7960 else:
7961 base_font_size = size_col.number_input(
7962 "Plot font size (px)" if scale_text_to_boxes else "Font size (px)",
7963 min_value=6,
7964 max_value=72,
7965 step=1,
7966 key="global_base_font_size",
7967 persist_state="session",
7968 disabled=disabled,
7969 label_visibility="collapsed",
7970 )
7972 font = _sub_row(
7973 "Font",
7974 caption_help=tip(
7975 "The word labels' font: a font name (e.g. 'Courier New') or a CSS "
7976 "font stack. **Multilingual** fills in a stack for CJK, Hebrew and "
7977 "Arabic."
7978 ),
7979 )
7980 family_col, stack_col = font.columns(
7981 [0.6, 0.4], gap=_LABEL_GAP, vertical_alignment="center"
7982 )
7983 font_family = family_col.text_input(
7984 "Text font",
7985 key="global_font_family",
7986 persist_state="session",
7987 disabled=disabled,
7988 label_visibility="collapsed",
7989 )
7990 stack_col.button(
7991 "Multilingual",
7992 on_click=lambda: st.session_state.update(
7993 global_font_family=_MULTILINGUAL_FONT_STACK
7994 ),
7995 disabled=disabled,
7996 width="stretch",
7997 )
7999 # Seeded rather than given a `value=`: restored pre-widget by a deep
8000 # link / saved config (BUG-17). `seed_canvas_state` pins it too, so a
8001 # custom background survives the runs this picker is greyed.
8002 _pin("global_bg_custom", DEFAULT_BACKGROUND_COLOR)
8003 color = _sub_row(
8004 "Color",
8005 caption_help=tip("The reading text's color, and the plot background."),
8006 )
8007 text_color_col, bg_cap_col, bg_col, bg_custom_col = color.columns(
8008 [0.17, 0.33, 0.33, 0.17], gap=_LABEL_GAP, vertical_alignment="center"
8009 )
8010 text_color_col.color_picker(
8011 "Text color",
8012 key="global_text_color",
8013 persist_state="session",
8014 disabled=disabled,
8015 label_visibility="collapsed",
8016 )
8017 _sub_caption(bg_cap_col, "Background")
8018 bg_choice = bg_col.selectbox(
8019 "Plot background",
8020 options=list(BACKGROUND_PRESETS.keys()) + ["Custom…"],
8021 key="global_bg_choice",
8022 persist_state="session",
8023 disabled=disabled,
8024 label_visibility="collapsed",
8025 )
8026 bg_custom_col.color_picker(
8027 "Custom background color",
8028 key="global_bg_custom",
8029 persist_state="session",
8030 disabled=disabled or bg_choice != "Custom…",
8031 label_visibility="collapsed",
8032 )
8034 if "right_to_left" in words_filtered and words_filtered["right_to_left"].any():
8035 st.caption(
8036 "↔ Right-to-left script detected — labels are laid out by the browser."
8037 )
8038 hint = _stimulus_font_install_hint(font_css)
8039 if hint is not None:
8040 font_name, font_url = hint
8041 st.caption(
8042 f"{ICONS['info']} This corpus was rendered in **{font_name}**. For "
8043 "the overlaid text to match the stimulus image exactly, install "
8044 "that font (it isn't bundled), then reload — otherwise labels "
8045 f"(especially URLs / Latin) can drift. [Download]({font_url}). Or "
8046 "turn on the stimulus **Image** to read the original text."
8047 )
8048 line_spacing_value = (
8049 float(line_spacing)
8050 if line_spacing is not None
8051 else float(st.session_state.get("global_line_spacing", seeded[4]))
8052 )
8053 return (
8054 int(base_font_size),
8055 str(font_family),
8056 line_spacing_value,
8057 bool(scale_text_to_boxes),
8058 )
8061def render_canvas_controls(
8062 words_filtered: pd.DataFrame,
8063 fixations_filtered: pd.DataFrame,
8064 data_choice: str | None = None,
8065 slot=None,
8066 expanded: bool = False,
8067 title: str = "Experimental Setup",
8068 bare: bool = False,
8069 text_host=None,
8070 render_text: bool = True,
8071 text_disabled: bool = False,
8072 text_section: str | None = None,
8073) -> tuple[int, int, int, str, float, bool]:
8074 """Render the canvas-geometry, typography and background panel.
8076 These controls let the user match the visualization to the experimental
8077 display, which is what keeps coordinates and word boxes spatially accurate.
8079 The panel normally renders into ``slot`` as its own collapsible expander.
8080 Pass ``bare=True`` when ``slot`` is already the disclosure container, as the
8081 compact Scanpath rail does. The setup wizard keeps the standalone expander.
8082 `seed_canvas_state` does the state work and is called first here, so rendering
8083 and not-rendering resolve identically.
8085 **UX-80/81 — ``text_host`` is where the typography half goes.** The two
8086 halves answer different questions and, in the rail, now live in different
8087 sections: the **screen** half (the monitor the figure is framed on) is drawn
8088 into ``slot`` for 📐 Figure & canvas, and the **text** half (how the reading
8089 text is drawn) into ``text_host`` for 📄 Stimulus → *Text*, beside the layer
8090 it describes. One call renders both — a widget drawn twice is a duplicate-key
8091 error, and one not drawn at all loses its key at the end of the run. The
8092 wizard passes neither and gets both, flat, in one expander: that step *is*
8093 the setup form, and hiding half of it behind buttons would be a step you
8094 cannot read at a glance.
8096 **The physical-geometry controls are not drawn in ``bare`` mode** (UX-81).
8097 Monitor physical width, viewing distance and display DPI are experiment
8098 facts, and the 🗂️ Data page's **Recording setup** (#DATA-22) already asks for
8099 them with a provenance; a second set in the rail could disagree with it. The
8100 *values* are unaffected — ``seed_canvas_state`` still pins them, and every
8101 consumer (px/degree for saccade amplitude in degrees, the point-to-pixel
8102 stimulus font conversion) reads them from state exactly as before, which is
8103 also what keeps a share link carrying them working.
8105 **UX-163/164 — in ``bare`` mode the rows take the rail popovers' shape.**
8106 The monitor's width and height share one ``Monitor | W × H px`` row, and the
8107 typography is four captioned rows — *Fit*, *Size*, *Font*, *Color* — under
8108 the 📄 Stimulus popover's *Text* row (`_rail_text_rows`). ``text_disabled``
8109 greys them while the Text layer is off, rather than leaving them undrawn;
8110 ``text_section`` titles the group where no *Text* row precedes it (the
8111 Corpus figure-style panel). The wizard's standalone form is unchanged.
8113 Returns:
8114 Tuple of (canvas_width, canvas_height, base_font_size, font_family,
8115 line_spacing, scale_text_to_boxes).
8116 """
8117 seeded = seed_canvas_state(words_filtered, fixations_filtered, data_choice)
8118 _, font_css = _dataset_font(words_filtered)
8119 host = slot if slot is not None else st.container()
8120 display = host if bare else host.expander(title, expanded=expanded)
8122 def field(host, kind: str, label: str, **kwargs):
8123 """One control, `label | field` in the rail and label-above in the wizard.
8125 UX-51 made the rail read as a compact form; the wizard's flat expander
8126 keeps the label above its field, for the same reason it stays flat — that
8127 step *is* the setup form, laid out across the page rather than inside a
8128 28rem popover, so there is no height to buy back.
8129 """
8130 if bare:
8131 return _labeled(host, kind, label, **kwargs)
8132 kwargs.pop("display", None)
8133 return getattr(host, kind)(label, **kwargs)
8135 # UX-80: no sub-popovers any more — each half is drawn straight into the
8136 # section that owns it, and the caller has already opened the one disclosure.
8137 screen = display
8138 text = text_host if (bare and text_host is not None) else display
8139 if bare:
8140 # The monitor's size is the dataset's Recording setup, set on the 🗂️ Data
8141 # page; the rail only frames the figure on it. `seed_canvas_state` has
8142 # resolved both keys above, and no widget owns them here, so nothing can
8143 # drop them at the end of a run.
8144 canvas_width, canvas_height = int(seeded[0]), int(seeded[1])
8145 else:
8146 canvas_width = field(
8147 screen,
8148 "number_input",
8149 "Monitor width (px)",
8150 min_value=100,
8151 max_value=10000,
8152 step=10,
8153 help="Use the real monitor width in pixels to keep coordinates true "
8154 "to scale.",
8155 key="global_canvas_width",
8156 persist_state="session",
8157 )
8158 canvas_height = field(
8159 screen,
8160 "number_input",
8161 "Monitor height (px)",
8162 min_value=100,
8163 max_value=10000,
8164 step=10,
8165 help="Use the real monitor height in pixels to keep coordinates true "
8166 "to scale.",
8167 key="global_canvas_height",
8168 persist_state="session",
8169 )
8170 # DATA-2: physical setup values live beside the pixel canvas they explain.
8171 # They are persisted with the plot config and immediately yield a px/degree
8172 # scale for downstream saccade/reporting work.
8173 #
8174 # **UX-81 — in the rail these three are not drawn at all.** They are
8175 # experiment facts, and the 🗂️ Data page's Recording setup (#DATA-22) already
8176 # asks for them *with a provenance*; a second set here could disagree with
8177 # it, and did. Not rendering a widget normally loses its key — but these keys
8178 # are not widget-owned any more: `seed_canvas_state` pins all three on every
8179 # run (including the derived DPI), so a share link or saved config still
8180 # restores them and every consumer reads the same numbers as before.
8181 # The wizard's standalone form still shows them: that *is* where they are set.
8182 if bare:
8183 monitor_width_mm = float(st.session_state.get("global_monitor_width_mm", 597.0))
8184 display_dpi = float(st.session_state.get("global_display_dpi", 96.0))
8185 else:
8186 monitor_width_mm = field(
8187 screen,
8188 "number_input",
8189 "Monitor physical width (mm)",
8190 display="Physical width (mm)",
8191 min_value=100.0,
8192 max_value=3000.0,
8193 step=1.0,
8194 key="global_monitor_width_mm",
8195 persist_state="session",
8196 help="Width of the visible display area, not the diagonal size.",
8197 )
8198 field(
8199 screen,
8200 "number_input",
8201 "Viewing distance (mm)",
8202 min_value=100.0,
8203 max_value=3000.0,
8204 step=10.0,
8205 key="global_viewing_distance_mm",
8206 persist_state="session",
8207 help="Eye-to-screen distance during the experiment.",
8208 )
8209 derived_dpi = float(canvas_width) / (float(monitor_width_mm) / 25.4)
8210 display_dpi = field(
8211 screen,
8212 "number_input",
8213 "Display DPI",
8214 min_value=20.0,
8215 max_value=1000.0,
8216 step=1.0,
8217 key="global_display_dpi",
8218 persist_state="session",
8219 help="Used for point-to-pixel stimulus font conversion. The physical "
8220 f"width above implies {derived_dpi:.1f} DPI.",
8221 )
8222 # No derived visual-angle figure (px/degree) is shown anywhere: nothing in
8223 # this release draws in degrees.
8225 # Text can be switched off while this function still supplies the screen
8226 # half to 📐 Figure & canvas. Before BUG-38, the caller passed an undefined
8227 # text container in that state and the whole Scanpath view crashed. Do not
8228 # move the typography controls into the Figure popover as a fallback: they
8229 # belong to Stimulus → Text, and their session-persistent keys already keep
8230 # the last values while that layer is hidden.
8231 if not render_text:
8232 return (
8233 int(canvas_width),
8234 int(canvas_height),
8235 int(seeded[2]),
8236 str(seeded[3]),
8237 float(seeded[4]),
8238 bool(seeded[5]),
8239 )
8241 if bare:
8242 base_font_size, font_family, line_spacing, scale_text_to_boxes = (
8243 _rail_text_rows(
8244 text,
8245 seeded=seeded,
8246 display_dpi=float(display_dpi),
8247 words_filtered=words_filtered,
8248 font_css=font_css,
8249 disabled=text_disabled,
8250 section=text_section,
8251 )
8252 )
8253 return (
8254 int(canvas_width),
8255 int(canvas_height),
8256 int(base_font_size),
8257 font_family,
8258 float(line_spacing),
8259 bool(scale_text_to_boxes),
8260 )
8262 # --- 🔤 Text & fonts (the wizard's flat form) -------------------------
8263 # Reading text is true-to-scale by default: it auto-sizes to the word boxes
8264 # (text height = box_height / line_spacing) and scales with the figure, so it
8265 # always fills the real line slot. Untick to fall back to a fixed font size.
8266 # Keyed (+ seeded) so the settings file can capture/reapply them.
8267 scale_text_to_boxes = field(
8268 text,
8269 "checkbox",
8270 "Scale text to boxes",
8271 key="global_scale_text_to_boxes",
8272 persist_state="session",
8273 help="Size text from word-box height. Without boxes, the plot font size "
8274 "is used instead.",
8275 )
8276 line_spacing = float(st.session_state.get("global_line_spacing", seeded[4]))
8277 use_stimulus_font_pt = bool(
8278 st.session_state.get("global_use_stimulus_font_pt", False)
8279 )
8280 stimulus_font_pt = float(st.session_state.get("global_stimulus_font_pt", 12.0))
8281 if scale_text_to_boxes:
8282 line_spacing = field(
8283 text,
8284 "number_input",
8285 "Line spacing",
8286 min_value=1.0,
8287 max_value=10.0,
8288 step=0.5,
8289 key="global_line_spacing",
8290 persist_state="session",
8291 help="Line slots represented by each word box. OneStop uses 3.",
8292 )
8293 else:
8294 use_stimulus_font_pt = field(
8295 text,
8296 "segmented_control",
8297 "Font unit",
8298 options=[False, True],
8299 format_func=lambda use_pt: "Points (pt)" if use_pt else "Pixels (px)",
8300 key="global_use_stimulus_font_pt",
8301 persist_state="session",
8302 help="Choose the original stimulus unit. Points are converted with "
8303 "the dataset DPI: px = pt × DPI ÷ 72.",
8304 )
8305 if use_stimulus_font_pt:
8306 stimulus_font_pt = field(
8307 text,
8308 "number_input",
8309 "Font size (pt)",
8310 min_value=4.0,
8311 max_value=144.0,
8312 step=0.5,
8313 key="global_stimulus_font_pt",
8314 persist_state="session",
8315 )
8316 st.session_state["global_base_font_size"] = int(
8317 min(max(round(font_pt_to_px(stimulus_font_pt, display_dpi)), 6), 72)
8318 )
8320 if not scale_text_to_boxes and use_stimulus_font_pt:
8321 base_font_size = int(st.session_state["global_base_font_size"])
8322 else:
8323 base_font_size = field(
8324 text,
8325 "number_input",
8326 "Plot font size (px)" if scale_text_to_boxes else "Font size (px)",
8327 min_value=6,
8328 max_value=72,
8329 step=1,
8330 help=(
8331 "Axis, legend, and fallback text size."
8332 if scale_text_to_boxes
8333 else "Reading-text, axis, and legend size in monitor pixels."
8334 ),
8335 key="global_base_font_size",
8336 persist_state="session",
8337 )
8338 text.button(
8339 "Use multilingual font stack",
8340 on_click=lambda: st.session_state.update(
8341 global_font_family=(
8342 "'Noto Sans', 'Noto Sans Hebrew', 'Noto Sans Arabic', "
8343 "'Noto Sans CJK SC', 'Arial Unicode MS', sans-serif"
8344 )
8345 ),
8346 help="A CJK/Hebrew/Arabic-capable CSS fallback stack.",
8347 )
8348 font_family = field(
8349 text,
8350 "text_input",
8351 "Text font",
8352 key="global_font_family",
8353 persist_state="session",
8354 help="Font for the word labels. Use the exact font from your experiment "
8355 "(e.g. 'Courier New') or a CSS fallback stack.",
8356 )
8357 if "right_to_left" in words_filtered and words_filtered["right_to_left"].any():
8358 text.caption(
8359 "↔ Right-to-left script detected — labels are laid out by the browser."
8360 )
8361 # When the dataset declares its stimulus typeface (MultiplEYE), the overlaid
8362 # text only lines up with the stimulus image if that exact font is installed
8363 # on the viewer's machine — we don't bundle it, and the browser otherwise
8364 # falls back per-script (CJK lands, but half-width Latin in a CJK font drifts,
8365 # e.g. URLs render too wide). Tell the user the font + how to get it.
8366 hint = _stimulus_font_install_hint(font_css)
8367 if hint is not None:
8368 font_name, font_url = hint
8369 text.caption(
8370 f"{ICONS['info']} This corpus was rendered in **{font_name}**. For the overlaid text "
8371 "to match the stimulus image exactly, install that font on this "
8372 "computer (it isn't bundled), then reload — otherwise the browser "
8373 "substitutes a fallback and labels (especially URLs / Latin) can "
8374 f"drift. [Download]({font_url}); install via Font Book (macOS), "
8375 "right-click → Install (Windows), or `~/.local/share/fonts` + "
8376 "`fc-cache -f` (Linux). Or just turn on the **stimulus image** to read "
8377 "the original text."
8378 )
8380 # Base reading-text colour (highlighted-text colour lives in Visualization
8381 # controls). Read back into viz_settings by controls.render_plot_controls.
8382 field(
8383 text,
8384 "color_picker",
8385 "Text color",
8386 key="global_text_color",
8387 persist_state="session",
8388 help="Colour of the reading text drawn over the stimulus.",
8389 )
8391 # Plot background lives here (Experimental Setup) rather than under
8392 # Visualization; render_plot_controls reads the chosen value from session state.
8393 bg_options = list(BACKGROUND_PRESETS.keys()) + ["Custom…"]
8394 field(
8395 text,
8396 "selectbox",
8397 "Plot background",
8398 options=bg_options,
8399 key="global_bg_choice",
8400 persist_state="session",
8401 help="Background of the plotting area (and exported figures).",
8402 )
8403 if st.session_state.get("global_bg_choice") == "Custom…":
8404 # Seed rather than pass `value=`: this key is restored pre-widget by a
8405 # deep link / saved config, and a keyed widget given both logs Streamlit's
8406 # "default value but also had its value set" warning (BUG-17).
8407 # This picker exists only while the choice is "Custom…", so it typically
8408 # FIRST mounts on a later run — the BUG-15 case, now handled by the
8409 # widget's own `persist_state="session"` (ENG-36) rather than by
8410 # re-asserting the value from Python on every run.
8411 _pin("global_bg_custom", DEFAULT_BACKGROUND_COLOR)
8412 field(
8413 text,
8414 "color_picker",
8415 "Custom background color",
8416 display="Custom color",
8417 key="global_bg_custom",
8418 persist_state="session",
8419 )
8421 return (
8422 int(canvas_width),
8423 int(canvas_height),
8424 int(base_font_size),
8425 font_family,
8426 float(line_spacing),
8427 bool(scale_text_to_boxes),
8428 )
8431# -----------------------------------------------------------------------------
8432# Setup wizard (hybrid: main-area on first load → collapsed panel afterward)
8433# -----------------------------------------------------------------------------
8436def _render_authoring_source() -> tuple[pd.DataFrame, pd.DataFrame]:
8437 """Standalone editor whose result joins the ordinary plot/export pipeline."""
8438 from scanpath_studio.authoring import (
8439 DEFAULT_LAYOUT,
8440 apply_authoring_event,
8441 authored_fixations,
8442 authoring_json,
8443 default_events,
8444 destructive_change,
8445 event_problems,
8446 layout_problems,
8447 layout_text,
8448 parse_authoring_document,
8449 reconcile_event_table,
8450 stale_target_words,
8451 unresolved_targets,
8452 unusable_event_rows,
8453 )
8454 from scanpath_studio.authoring_component import render_authoring_canvas
8456 source = st.session_state.get("data_source_choice", AUTHOR_CHOICE)
8457 drafts = st.session_state.setdefault("_manual_scanpath_drafts", {})
8458 previous = st.session_state.get("_author_editor_source")
8459 if previous != source:
8460 document = drafts.get(source)
8461 if document is None and (
8462 source == MANUAL_SAMPLE_CHOICE or previous is not None
8463 ):
8464 seed_text = (
8465 _MANUAL_SAMPLE_TEXT
8466 if source == MANUAL_SAMPLE_CHOICE
8467 else "Reading unfolds through a sequence of careful eye movements."
8468 )
8469 document = (
8470 seed_text,
8471 dict(DEFAULT_LAYOUT),
8472 default_events(layout_text(seed_text)),
8473 )
8474 if document is not None:
8475 seed_text, seed_layout, seed_events = document
8476 st.session_state["author_text"] = seed_text
8477 st.session_state["_author_layout"] = seed_layout
8478 st.session_state["_authored_events_frame"] = seed_events
8479 st.session_state["_author_text_for_events"] = seed_text
8480 st.session_state["_author_selected_fixation"] = None
8481 st.session_state["_author_events_editor_revision"] = (
8482 int(st.session_state.get("_author_events_editor_revision", 0)) + 1
8483 )
8484 st.session_state["_author_editor_source"] = source
8485 header = st.container(horizontal=True, vertical_alignment="center")
8486 header.subheader(
8487 f"{ICONS['author']} "
8488 + (
8489 "Edit the hand-drawn sample"
8490 if source == MANUAL_SAMPLE_CHOICE
8491 else "Author a scanpath"
8492 )
8493 )
8494 header.button("Cancel", key="cancel_authoring", on_click=_cancel_authoring)
8495 st.caption(
8496 "Write the stimulus, then click or drag on the canvas to place "
8497 "fixations. A fixation is drawn at its X/Y; its optional target word "
8498 "never moves it. The **Fixation table** edits the same fixations from "
8499 "the keyboard."
8500 )
8501 restored = st.file_uploader(
8502 "Restore authoring file",
8503 type=["json"],
8504 key=f"author_restore_upload_{source}",
8505 help="Load a JSON file previously saved from this editor.",
8506 max_upload_size=upload_limit_mb(),
8507 )
8508 if restored is not None:
8509 identity = (source, restored.name, restored.size)
8510 if st.session_state.get("_author_restore_identity") != identity:
8511 try:
8512 document = parse_authoring_document(restored.getvalue().decode("utf-8"))
8513 except (ValueError, UnicodeDecodeError) as exc:
8514 st.error(f"Couldn't restore this file: {exc}")
8515 else:
8516 # The draft it replaces becomes the previous draft at the end
8517 # of this run, so Restore previous draft brings it back.
8518 _load_author_draft(
8519 source, (document.text, document.layout, document.events)
8520 )
8521 st.session_state["_author_restore_identity"] = identity
8523 st.session_state.setdefault(
8524 "author_text", "Reading unfolds through a sequence of careful eye movements."
8525 )
8526 text = st.text_area(
8527 "Stimulus text",
8528 key="author_text",
8529 persist_state="session",
8530 height=100,
8531 )
8532 layout = {**DEFAULT_LAYOUT, **st.session_state.get("_author_layout", {})}
8533 words = layout_text(text, **layout)
8534 with st.expander("Parsed word geometry", expanded=False):
8535 if words.empty:
8536 st.info(
8537 "Enter stimulus text to create word boxes. The empty canvas is still valid."
8538 )
8539 else:
8540 line_count = int(words["line_idx"].max()) + 1
8541 st.caption(
8542 f"{plural(len(words), 'word')} across {plural(line_count, 'line')} "
8543 "· words are numbered from 1, lines from 0; blank lines are kept."
8544 )
8545 st.dataframe(
8546 words[["text", "word_id", "line_idx", "x", "y", "width", "height"]],
8547 hide_index=True,
8548 width="stretch",
8549 )
8550 for problem in layout_problems(
8551 words, canvas_width=layout["canvas_width"], margin=layout["margin"]
8552 ):
8553 st.warning(problem)
8555 events_text = st.session_state.get("_author_text_for_events")
8556 if events_text is None:
8557 # A fresh editor: one fixation per word to start from.
8558 st.session_state.setdefault("_authored_events_frame", default_events(words))
8559 st.session_state.setdefault("_author_events_editor_revision", 0)
8560 st.session_state["_author_text_for_events"] = text
8561 elif events_text != text:
8562 # Editing the text keeps every authored fixation — id, X/Y, order and
8563 # duration. Only a target word can go out of date, and that is flagged
8564 # below rather than regenerated. The current table is the base for the
8565 # flag: `_authored_events_frame` lags behind the table's own edits.
8566 previous_draft = drafts.get(source)
8567 current = previous_draft[2] if previous_draft else None
8568 stale = st.session_state.setdefault(_AUTHOR_STALE_TARGETS_KEY, {}).setdefault(
8569 source, {}
8570 )
8571 found = stale_target_words(layout_text(events_text, **layout), words, current)
8572 # An entry already flagged keeps the word it named first, so undoing
8573 # the edit (or fixing it in two steps) resolves it.
8574 for fixation_id, entry in found.items():
8575 stale.setdefault(fixation_id, entry)
8576 st.session_state["_author_text_for_events"] = text
8577 seed = st.session_state.get("_authored_events_frame", default_events(words))
8578 last_word = int(words["word_id"].max()) if not words.empty else 1
8579 canvas_panel = st.container()
8580 fixation_table = st.expander("Fixation table", expanded=False)
8581 fixation_table.caption(
8582 "One row per fixation. **Fixation id** is stable; **Order** controls the "
8583 "reading sequence. X/Y place the marker in screen pixels. **Target word** "
8584 f"is optional (1–{last_word}) and may be edited independently; blank X/Y "
8585 "fall back to that word's center."
8586 )
8587 # BUG-19: the editor's own key holds the edits as a delta against `seed`, so
8588 # `seed` must stay the STABLE base — it is reseeded only when the stimulus
8589 # text changes or a file is restored. Writing the returned frame back into it
8590 # applies that delta twice, and after a row deletion leaves a gapped index,
8591 # which `num_rows="dynamic"` cannot add rows to: from there edits land on the
8592 # wrong rows and rows disappear. Read the edits from the return value only.
8593 editor_revision = int(st.session_state.get("_author_events_editor_revision", 0))
8594 events = fixation_table.data_editor(
8595 seed,
8596 key=f"author_events_editor_{editor_revision}",
8597 num_rows="dynamic",
8598 hide_index=True,
8599 column_config={
8600 "fixation_id": st.column_config.NumberColumn(
8601 "Fixation id",
8602 disabled=True,
8603 help="The fixation's fixed number, shared by the canvas and this table.",
8604 ),
8605 "order_in_trial": st.column_config.NumberColumn(
8606 "Order",
8607 min_value=1,
8608 step=1,
8609 help="Reading order. Each fixation needs a unique whole number.",
8610 ),
8611 "word_id": st.column_config.NumberColumn(
8612 "Target word (optional)",
8613 min_value=1,
8614 max_value=last_word,
8615 step=1,
8616 help=(
8617 "Optional: the word this fixation belongs to, counting from 1. "
8618 "It does not move the marker."
8619 ),
8620 ),
8621 "x": st.column_config.NumberColumn(
8622 "X (px)",
8623 help="Horizontal screen coordinate; independent of target word.",
8624 ),
8625 "y": st.column_config.NumberColumn(
8626 "Y (px)", help="Vertical screen coordinate; independent of target word."
8627 ),
8628 "duration_ms": st.column_config.NumberColumn(
8629 "Duration (ms)",
8630 min_value=1,
8631 help="How long the fixation lasts. It also sets the marker size.",
8632 ),
8633 },
8634 width="stretch",
8635 )
8636 selected = st.session_state.get("_author_selected_fixation")
8637 try:
8638 effective_events, selected = reconcile_event_table(events, selected)
8639 except ValueError as exc:
8640 st.error(
8641 f"Fix the Fixation table before these edits can be drawn or saved: {exc}"
8642 )
8643 effective_events, selected = reconcile_event_table(seed, selected)
8644 events_valid = False
8645 else:
8646 events_valid = True
8647 st.session_state["_author_selected_fixation"] = selected
8649 for problem in event_problems(words, events):
8650 if not problem.startswith(("Fixation id", "Order")):
8651 st.warning(problem)
8652 dropped = unusable_event_rows(words, events)
8653 if dropped:
8654 st.caption(
8655 "Rows with no X/Y and no valid target word aren't drawn, and block "
8656 "**Save dataset**."
8657 )
8658 stale_entries = st.session_state.get(_AUTHOR_STALE_TARGETS_KEY, {}).get(source, {})
8659 stale = unresolved_targets(stale_entries, words, effective_events)
8660 if stale_entries and len(stale) < len(stale_entries):
8661 # Resolved ones (target changed, fixation deleted, text put back) go.
8662 st.session_state[_AUTHOR_STALE_TARGETS_KEY][source] = {
8663 fixation_id: stale_entries[fixation_id] for fixation_id in stale
8664 }
8665 if stale:
8666 listed = ", ".join(
8667 f"fixation {fixation_id} → word {word_id}"
8668 for fixation_id, word_id in sorted(stale.items())[:8]
8669 )
8670 more = f" (+{len(stale) - 8} more)" if len(stale) > 8 else ""
8671 st.warning(
8672 f"The text edit changed or removed the target word of "
8673 f"{len(stale)} {'fixation' if len(stale) == 1 else 'fixations'}: "
8674 f"{listed}{more}. Their position, timing and order are unchanged. "
8675 "Set each **Target word** in the Fixation table, or clear them.",
8676 icon=ICONS["warning"],
8677 )
8678 if st.button("Clear those target words", key=f"author_clear_stale_{source}"):
8679 cleared = effective_events.copy()
8680 mask = cleared["fixation_id"].map(int).isin(stale)
8681 cleared["word_id"] = cleared["word_id"].astype(object)
8682 cleared.loc[mask, "word_id"] = None
8683 st.session_state["_authored_events_frame"] = cleared
8684 st.session_state["_author_events_editor_revision"] = editor_revision + 1
8685 st.session_state[_AUTHOR_STALE_TARGETS_KEY].pop(source, None)
8686 st.rerun()
8688 canvas_height = max(
8689 480,
8690 int(words["y"].max() + words["height"].max() + layout["margin"])
8691 if not words.empty
8692 else 480,
8693 )
8694 with canvas_panel:
8695 canvas_event = render_authoring_canvas(
8696 words,
8697 effective_events,
8698 canvas_width=layout["canvas_width"],
8699 canvas_height=canvas_height,
8700 selected_fixation_id=selected,
8701 )
8702 if canvas_event:
8703 try:
8704 updated, selected = apply_authoring_event(
8705 effective_events,
8706 canvas_event,
8707 selected_fixation_id=selected,
8708 )
8709 except ValueError as exc:
8710 st.error(f"That canvas edit couldn't be applied: {exc}")
8711 else:
8712 st.session_state["_authored_events_frame"] = updated
8713 st.session_state["_author_selected_fixation"] = selected
8714 # A data editor's browser delta is keyed against the frame it first
8715 # mounted with. Give canvas-authored data a fresh key so the visible
8716 # table mounts from the new coordinates instead of retaining its old
8717 # client-side base (the same invariant as BUG-19, in the other
8718 # direction).
8719 st.session_state["_author_events_editor_revision"] = editor_revision + 1
8720 st.rerun()
8722 fixations = authored_fixations(words, effective_events)
8723 name_key = f"author_dataset_name_{source}"
8724 st.text_input(
8725 "Dataset name",
8726 value="Hand-drawn sample (edited)"
8727 if source == MANUAL_SAMPLE_CHOICE
8728 else "My scanpath",
8729 key=name_key,
8730 required=True,
8731 # Streamlit 1.65: read only by the Save dataset callback.
8732 on_change="ignore",
8733 )
8734 can_save = events_valid and not words.empty and not fixations.empty and not dropped
8735 if can_save:
8736 st.session_state["_author_save_payload"] = {
8737 "words": words,
8738 "fixations": fixations,
8739 "raw_gaze": pd.DataFrame(),
8740 "authoring": authoring_json(text, effective_events, layout=layout),
8741 }
8742 else:
8743 st.session_state.pop("_author_save_payload", None)
8744 actions = st.container(horizontal=True, vertical_alignment="center")
8745 actions.button(
8746 "Save dataset",
8747 icon=ICONS["save"],
8748 key="save_authored_dataset",
8749 type="primary",
8750 disabled=not can_save,
8751 on_click=_save_authored_dataset,
8752 args=(name_key,),
8753 )
8754 actions.download_button(
8755 "Download authoring file",
8756 data=authoring_json(text, effective_events, layout=layout),
8757 file_name=AUTHORING_FILE_NAME,
8758 mime="application/json",
8759 icon=ICONS["download"],
8760 key=f"author_download_{source}",
8761 on_click="ignore",
8762 disabled=not events_valid,
8763 help=(
8764 "The editable draft — text, layout and every fixation with its id, "
8765 "position, order and duration. Load it again with **Restore "
8766 "authoring file**, or from a script with `load_authored_scanpath`."
8767 ),
8768 )
8769 previous = st.session_state.get(_AUTHOR_PREVIOUS_DRAFT_KEY, {}).get(source)
8770 actions.button(
8771 "Restore previous draft",
8772 icon=ICONS["undo"],
8773 key=f"author_restore_previous_{source}",
8774 disabled=previous is None,
8775 on_click=_restore_previous_author_draft,
8776 args=(source,),
8777 help=(
8778 "Go back to the draft before the last change that removed, moved or "
8779 "retimed fixations — one step. Press again to return."
8780 ),
8781 )
8782 actions.button(
8783 "Reset to one fixation per word",
8784 key=f"author_reset_fixations_{source}",
8785 disabled=words.empty,
8786 on_click=_reset_author_fixations,
8787 args=(source, words),
8788 help=(
8789 "Replace every fixation with one per word, 220 ms each. "
8790 "**Restore previous draft** brings the current ones back."
8791 ),
8792 )
8793 draft = (text, dict(layout), effective_events.copy())
8794 last = drafts.get(source)
8795 if events_valid and last is not None and destructive_change(last[2], draft[2]):
8796 st.session_state.setdefault(_AUTHOR_PREVIOUS_DRAFT_KEY, {})[source] = last
8797 drafts[source] = draft
8798 return words, fixations
8801#: PRE-1 control defaults. These keys are a wire format — a deep link or a saved
8802#: config writes them into session state via `url_state` *before* the widgets
8803#: render, so the widgets must NOT also pass `value=`: Streamlit warns when a
8804#: keyed widget is given both (BUG-17). A plain `setdefault` is enough here (the
8805#: expander's contents render every run, so these never mount late — unlike the
8806#: `persist_state` cases above); see `controls._pin` for that distinction.
8807#: Keep in sync with the fallbacks in `url_state._restore_plot_config` and
8808#: `tabs._build_studio_config`.
8809_PREPROC_DEFAULTS: dict = {
8810 "global_preproc_enabled": False,
8811 "global_preproc_short_policy": "Off",
8812 "global_preproc_short_threshold_ms": 80.0,
8813 "global_preproc_merge_distance_chars": 1.0,
8814 "global_preproc_blink_adjacent": True,
8815}
8817#: PRE-22 — what `_preprocessing_settings` answers while the feature is hidden:
8818#: the shape every caller expects, with the pipeline off. Mirrors
8819#: ``_PREPROC_DEFAULTS`` in values, but keyed the way the settings dict is.
8820_PREPROC_SETTINGS_OFF = {
8821 "enabled": False,
8822 "short_policy": "Off",
8823 "short_threshold_ms": 80.0,
8824 "merge_distance_chars": 1.0,
8825 "discard_blink_adjacent": True,
8826}
8829def _preprocessing_settings(host=None) -> dict:
8830 """Render the PRE-1 controls and return their cache-key-safe settings.
8832 ``host`` is the 🧹 Preprocessing section of the Data page (DATA-26). Renders
8833 bare into it — the section heading is the page's, written into the slot above
8834 these widgets by ``main``.
8836 It belongs on the page rather than in the Scanpath rail (where UX-38 floated
8837 putting it) because ``app._preprocessing_settings`` reshapes the frames
8838 *every* view reads, including Corpus Analysis, which has no rail: a
8839 dataset-wide control must not be unreachable from one of the views it
8840 changes.
8842 **PRE-22**: while the feature is held back from the release, nothing renders
8843 and the returned settings are the "off" ones — whatever session state holds.
8844 A saved config or share link from a build that *did* show the panel carries
8845 `global_preproc_*` values, and honouring them would run a pipeline with no
8846 control anywhere to see it, undo it, or explain the changed numbers. They are
8847 ignored rather than cleared, so re-enabling the panel finds them intact.
8848 """
8849 if not preprocessing_enabled():
8850 return dict(_PREPROC_SETTINGS_OFF)
8851 apply_pending_preprocessing()
8852 for key, default in _PREPROC_DEFAULTS.items():
8853 st.session_state.setdefault(key, default)
8854 with host if host is not None else st.container():
8855 enabled = st.toggle(
8856 "Enable preprocessing",
8857 key="global_preproc_enabled",
8858 persist_state="session",
8859 help="Optional and off by default. Original rows remain available; "
8860 "excluded rows are soft-marked with a reason.",
8861 )
8862 policy = st.selectbox(
8863 "Short fixations",
8864 ["Off", "Merge", "Merge then discard", "Discard"],
8865 key="global_preproc_short_policy",
8866 persist_state="session",
8867 disabled=not enabled,
8868 )
8869 threshold = st.number_input(
8870 "Short threshold (ms)",
8871 min_value=1.0,
8872 max_value=500.0,
8873 key="global_preproc_short_threshold_ms",
8874 persist_state="session",
8875 disabled=not enabled or policy == "Off",
8876 )
8877 distance = st.number_input(
8878 "Merge distance (characters)",
8879 min_value=0.25,
8880 max_value=10.0,
8881 step=0.25,
8882 key="global_preproc_merge_distance_chars",
8883 persist_state="session",
8884 disabled=not enabled or "Merge" not in policy,
8885 )
8886 blink = st.toggle(
8887 "Exclude blink-adjacent fixations",
8888 key="global_preproc_blink_adjacent",
8889 persist_state="session",
8890 disabled=not enabled,
8891 )
8892 if st.button("Recompute preprocessing", disabled=not enabled):
8893 st.cache_data.clear()
8894 return {
8895 "enabled": enabled,
8896 "short_policy": policy,
8897 "short_threshold_ms": threshold,
8898 "merge_distance_chars": distance,
8899 "discard_blink_adjacent": blink,
8900 }
8903def _activate_data_source(data_choice: str, *, preproc_host=None) -> dict:
8904 """Reset source-scoped state and return the active preprocessing settings.
8906 ``preproc_host`` is the 🧹 Preprocessing section of the Data page (DATA-26 —
8907 it was a popover on the top menu bar until the page took it).
8908 """
8909 st.session_state["_active_data_source"] = data_choice
8910 if st.session_state.get(_AUTHOR_EDITING_KEY) != data_choice:
8911 st.session_state.pop(_AUTHOR_EDITING_KEY, None)
8912 if not _authoring_editor_open(data_choice):
8913 st.session_state.pop("_author_editor_source", None)
8914 preprocessing = _preprocessing_settings(preproc_host)
8915 if st.session_state.get("_share_selection_source") != data_choice:
8916 st.session_state.pop("_share_selection", None)
8917 st.session_state["_share_selection_source"] = data_choice
8919 # The trial-filter stash is keyed per *corpus*: a filter narrowed to a value
8920 # only one corpus has (a text id, a reader) must not ride into the next one,
8921 # where no trial can satisfy it and nothing says why the pool went empty.
8922 # `public_dataset_choice` alone is enough for that again — every corpus,
8923 # prepared benchmark ones included, is its own registry entry — so the extra
8924 # `eyegenbench_dataset` component R31 needed is gone (DATA-27 Task 11R).
8925 source_key = (data_choice, st.session_state.get("public_dataset_choice"))
8926 if st.session_state.get("_filters_for") == source_key:
8927 return preprocessing
8929 previous = st.session_state.get("_filters_for")
8930 stash = st.session_state.setdefault("_filter_stash", {})
8931 if previous is not None:
8932 stash[previous] = dict(st.session_state.get("_trial_filters_raw", {}))
8933 stale_keys = [
8934 key
8935 for key in st.session_state
8936 if isinstance(key, str) and key.startswith("filter_")
8937 ]
8938 for key in stale_keys:
8939 del st.session_state[key]
8940 st.session_state.pop("_trial_filters", None)
8941 restored = stash.get(source_key)
8942 st.session_state["_trial_filters_raw"] = dict(restored) if restored else {}
8943 st.session_state["_filters_for"] = source_key
8944 return preprocessing
8947#: UX-168: the last dataset a run left on screen — where Cancel goes back to
8948#: (`_remember_open_dataset`). Not the wizard's `_prev_source`, which only
8949#: records where leaving the add-dataset wizard returns to.
8950LAST_LOADED_SOURCE_KEY = "_sps_last_loaded_source"
8951#: Every dataset this session has had on screen — the table's *Loaded* status.
8952#: Session-only on purpose: the remembered counts outlive a restart, the
8953#: loaders' caches do not.
8954LOADED_THIS_SESSION_KEY = "_sps_loaded_this_session"
8955#: UX-166: the dataset task this session's pipeline is running, set when the
8956#: card opens and cleared when it ends; found still set by the next run, it
8957#: means that run was abandoned mid-load.
8958DATASET_TASK_KEY = "_sps_dataset_task"
8959#: UX-168: set by Cancel on the dataset card, read once by the next run's notice.
8960CANCELLED_LOAD_KEY = "_sps_cancelled_load"
8963def _dataset_cancel(token: str, task_key: tuple) -> loading.Cancel | None:
8964 """The dataset card's Cancel: only for a load the user started — cold, or
8965 by switching away from what was already on screen.
8967 **T8-2:** the last dataset a run left on screen (`_remember_open_dataset`)
8968 is where Cancel goes back to, but a rerun of **that same dataset** (a filter
8969 change, a re-normalization, a slow step on ✏️ Edit dataset) must offer none
8970 at all — there is nothing the user asked to leave, and clicking it would
8971 abandon the dataset they are looking at. The Bundled demo is the fallback
8972 only when no dataset has been on screen yet this session, never back to
8973 `token` itself either way.
8975 **T8-5:** `back` must also be a dataset this run can actually open —
8976 `resolve_data_source` heals a stale/hidden/removed selection to
8977 `entries[0]` (published as `_data_source_entries`, before this runs), so a
8978 button still reading "Back to <back>" could silently reopen something else.
8979 Falls back to the demo when it is offered and isn't `token`, else no
8980 Cancel.
8981 """
8982 last = st.session_state.get(LAST_LOADED_SOURCE_KEY)
8983 if last == token:
8984 return None
8985 back = last or DEMO_CHOICE
8986 if back == token:
8987 return None
8988 entries = st.session_state.get("_data_source_entries") or ()
8989 if back not in entries:
8990 if DEMO_CHOICE not in entries or DEMO_CHOICE == token:
8991 return None
8992 back = DEMO_CHOICE
8993 return loading.Cancel(
8994 f"Back to {_dataset_display_name(back)}",
8995 _cancel_dataset_load,
8996 args=(task_key, token, _dataset_display_name(token), back),
8997 )
9000def _cancel_dataset_load(task_key: tuple, token: str, name: str, back: str) -> None:
9001 """UX-168: Cancel on the dataset card — stop the load, reopen ``back``."""
9002 progress.cancel(task_key)
9003 st.session_state["_pending_source_choice"] = back
9004 st.session_state[CANCELLED_LOAD_KEY] = {"token": token, "name": name}
9007def _retry_load(token: str) -> None:
9008 st.session_state["_pending_source_choice"] = token
9011def _render_cancelled_load_notice(host) -> None:
9012 """UX-168: "Stopped loading X · Try again", once, where the notices go."""
9013 note = st.session_state.pop(CANCELLED_LOAD_KEY, None)
9014 if not note:
9015 return
9016 row = host.container(
9017 key="sps_cancelled_notice", horizontal=True, vertical_alignment="center"
9018 )
9019 # As wide as its words, so Try again sits beside them, not across the page.
9020 row.caption(f"Stopped loading **{note['name']}**.", width="content")
9021 row.button(
9022 "Try again",
9023 key="sps_try_again",
9024 on_click=_retry_load,
9025 args=(note["token"],),
9026 type="tertiary",
9027 )
9030#: UX-199: ``"show"`` once the first upload lands on a deployment that keeps
9031#: nothing, ``"dismissed"`` once the user closes the reminder — which keeps it
9032#: down for the rest of the session, later uploads included. UI-only, so it is
9033#: not wire format and no link or saved file carries it.
9034BACKUP_REMINDER_KEY = "_sps_backup_reminder"
9036#: Where the docs list what to download, and from where, to keep your work.
9037BACKUP_GUIDE_URL = f"{CITATION['docs_url']}guides/outputs-sharing/#back-up-your-work"
9040def arm_backup_reminder() -> None:
9041 """Ask for the backup reminder after an upload, where nothing is saved (UX-199).
9043 Called from the wizard's ✅ Add dataset callback. *Saved on this computer*
9044 says the same thing at the foot of the 🗂️ Data page, where a hosted user
9045 working in Scanpath may never look — so the first upload, the moment there
9046 is something to lose, says it in the notices strip on every view.
9047 """
9048 if st.session_state.get(BACKUP_REMINDER_KEY) == "dismissed":
9049 return
9050 if persistence_enabled(str(getattr(st.context, "url", "") or "")):
9051 return
9052 st.session_state[BACKUP_REMINDER_KEY] = "show"
9055def _dismiss_backup_reminder() -> None:
9056 st.session_state[BACKUP_REMINDER_KEY] = "dismissed"
9059def _render_backup_reminder(host, active_view: str) -> None:
9060 """UX-199: the dismissible "nothing is saved here" reminder, in the notices."""
9061 if st.session_state.get(BACKUP_REMINDER_KEY) != "show":
9062 return
9063 box = host.container(key="sps_backup_reminder", border=True)
9064 box.markdown(
9065 f"{ICONS['warning']} **This deployment saves nothing.** Closing or "
9066 "refreshing the tab loses the datasets you added, their column mappings, "
9067 "your annotations and designs. Keep the files you uploaded, and export your "
9068 f"annotations from {ICONS['view_data']} **Data Management → Annotations** and each "
9069 "dataset's mapping from "
9070 f"{ICONS['edit']} **Edit dataset → Download setup file**. "
9071 f"[What to back up ↗]({BACKUP_GUIDE_URL})"
9072 )
9073 row = box.container(horizontal=True, gap="small")
9074 if active_view != _VIEW_DATA:
9075 row.button(
9076 "Go to Data Management",
9077 key="sps_backup_reminder_go",
9078 icon=ICONS["view_data"],
9079 on_click=_go_data,
9080 type="tertiary",
9081 )
9082 row.button(
9083 "Dismiss",
9084 key="sps_backup_reminder_dismiss",
9085 icon=ICONS["close"],
9086 on_click=_dismiss_backup_reminder,
9087 type="tertiary",
9088 )
9091def _open_dataset_card(
9092 page: loading.Page,
9093 data_choice: str,
9094 *,
9095 view: str,
9096 view_switched: bool,
9097 finalizing: bool,
9098) -> loading.Card | None:
9099 """UX-166: this run's dataset card, or ``None`` when there is no load to wait on.
9101 A switch to ``view`` opens the task-less "Opening <view>" card at once,
9102 titled for the view and without steps, **only when no load was already in
9103 flight** (``in_flight`` below): the dataset is loaded already, so the
9104 skeleton is what answers the click, and each switch's task is its own
9105 (never joined by a load's task, nor joining one — nor a later switch to the
9106 same view, which must never join a stale one either). **T8-3:** a view
9107 switch that instead lands on an in-flight load — the previous run's, or one
9108 a dataset pick in the same click just started — opens the ordinary dataset
9109 card instead (steps, Cancel, the explicit `task_key`, so it joins that
9110 load), revealed at once rather than task-less: a load's own progress and
9111 Cancel must never be hidden behind the view-switch skeleton. `DATASET_TASK_KEY`
9112 still names the dataset's either way — a view switch in the middle of a
9113 load is not "another dataset".
9115 **In flight** means a live, unfinished task still holds the key
9116 (`progress.running`), not just that the key is set: a run that ended by
9117 ``st.stop()``, an exception or ``st.rerun()`` mid-pipeline leaves
9118 `DATASET_TASK_KEY` behind with nothing computing it, and a view switch after
9119 that would otherwise show a load's card — steps, Cancel, revealed at once —
9120 for no load at all.
9122 **UX-168:** a run that finds *another* dataset's task still marked running
9123 under `DATASET_TASK_KEY` was started by picking this one mid-load (directly,
9124 or via Cancel) — the user changed their mind, so the abandoned load is
9125 cancelled here, at its next checkpoint, instead of competing for the CPU.
9126 Checked before the early return below too: switching to Upload or Author
9127 mid-load cancels a load left running just the same, and that branch has no
9128 load of its own to name `DATASET_TASK_KEY` after, so it clears the key
9129 rather than replacing it.
9131 **The load card is gated** (``reveal_on_work``): it opens on every run, and
9132 a plain rerun — every build a cache hit — can outlast the delay on a big
9133 corpus, re-hashing the frames each hit hands back; shown, it blanked the
9134 view to the skeleton on every widget touch. So it shows only once one of
9135 its builds has reported, i.e. missed. ``reveal_now`` still shows it at once.
9137 **A rerun of the dataset on screen is an update:** "Updating <dataset>",
9138 with no Cancel, no "last load" hint and no duration recorded — a filter
9139 change's re-run is not how long the dataset takes to open.
9140 """
9141 if data_choice == UPLOAD_CHOICE or _authoring_editor_open(data_choice):
9142 previous = st.session_state.pop(DATASET_TASK_KEY, None)
9143 if previous is not None:
9144 progress.cancel(tuple(previous))
9145 return None
9146 token = str(st.session_state.get("data_source_choice") or data_choice)
9147 session = loading.session_id()
9148 task_key = ("dataset", session, token)
9149 previous = st.session_state.get(DATASET_TASK_KEY)
9150 in_flight = previous is not None and progress.running(tuple(previous))
9151 if previous is not None and tuple(previous) != task_key:
9152 progress.cancel(tuple(previous))
9153 st.session_state[DATASET_TASK_KEY] = task_key
9154 if view_switched and not in_flight:
9155 return page.open_card(
9156 title=f"Opening {view_label(view)}",
9157 reveal_now=True,
9158 )
9159 stored = data_choice in st.session_state.get("_datasets", {})
9160 steps = (
9161 ("Building the trial list",)
9162 if stored
9163 else ("Reading files", "Mapping columns", "Building the trial list")
9164 )
9165 # UX-166: the dataset already on screen, run again — a filter change on a
9166 # big corpus, a re-normalization — is an update, not a load: titled so, with
9167 # no "last load" hint and no duration of its own, since an update's time
9168 # must never stand in for how long the dataset took to open. (It offers no
9169 # Cancel either — `_dataset_cancel`'s T8-2 rule, the same test.) A dataset
9170 # just added is always a load: the wizard, not a dataset, was on screen.
9171 updating = not finalizing and st.session_state.get(LAST_LOADED_SOURCE_KEY) == token
9172 return page.open_card(
9173 title=f"{'Updating' if updating else 'Loading'} {_dataset_display_name(token)}",
9174 steps=steps,
9175 cancel=_dataset_cancel(token, task_key),
9176 task_key=task_key,
9177 duration_key=None if updating else ("dataset", token),
9178 reveal_now=finalizing or view_switched,
9179 reveal_on_work=True,
9180 )
9183def _remember_open_dataset(data_choice: str) -> None:
9184 """UX-168: the dataset this run leaves on screen is where Cancel goes back to.
9186 Recorded on every path that leaves a dataset showing, not only a finished
9187 pipeline: the ✏️ Author editor (which opens no card), a dataset whose
9188 mapping still needs fixing, one whose filters left no trials — "Back to"
9189 must name the dataset you had open, not one further back. Never for the
9190 add-dataset wizard (``UPLOAD_CHOICE``), which shows no dataset and has its
9191 own way back (`_prev_source`). The value is the dataset card's own token.
9192 """
9193 if data_choice == UPLOAD_CHOICE:
9194 return
9195 token = str(st.session_state.get("data_source_choice") or data_choice)
9196 st.session_state[LAST_LOADED_SOURCE_KEY] = token
9197 st.session_state.setdefault(LOADED_THIS_SESSION_KEY, set()).add(token)
9200def _finish_dataset_card(card: loading.Card | None, data_choice: str) -> None:
9201 """UX-166: the pipeline finished — every step ✓, the skeleton left up — and
9202 the dataset is on screen (`_remember_open_dataset`), card or none."""
9203 if card is not None:
9204 card.finish()
9205 card.close(keep=True)
9206 st.session_state.pop(DATASET_TASK_KEY, None)
9207 _remember_open_dataset(data_choice)
9210def main() -> None:
9211 """Main application entry point: one script run, inside a loading scope.
9213 UX-165: ``loading.run_scope`` guarantees that no loading card's timer thread
9214 outlives the run that started it — whether the run ends normally, returns
9215 early, raises, or is abandoned by a click — and ends a run whose work was
9216 cancelled as a stopped one, which keeps the state of the widgets it never
9217 reached. The run itself is `_run_app`.
9218 """
9219 with loading.run_scope():
9220 _run_app()
9223def _run_app() -> None:
9224 """One script run, top to bottom (``main`` wraps it in a loading scope).
9226 1. Page setup: config and CSS, the URL presets, and the recovery-cache
9227 restore — under a card of its own until the session has restored.
9228 2. Chrome: the top nav resolves the active view; then the menu bar, the
9229 title and the tours and dialogs.
9230 3. Reserved slots, in screen order: the view area the Scanpath and Corpus
9231 views render into, then the 🗂️ Data page's overview and editor slots.
9232 4. The dataset pipeline, under the dataset card: load → normalize →
9233 filter → the trial list. It returns early, through ``_end_loading``, for
9234 an open add-dataset wizard, a mapping that can't be satisfied, or a
9235 filter that empties the pool.
9236 5. The active view: 🗺️ Scanpath or 📊 Corpus Analysis inside the view
9237 area, or the Data page's slots filled.
9238 6. The epilogue: the recovery-cache save, then the Data page's *Saved on
9239 this computer* section.
9240 """
9241 configure_page()
9242 # PERF-3: the cache-key memo is scoped to ONE script run — drop last run's
9243 # entries before anything fingerprints a frame, so a frame rebuilt this run
9244 # is hashed afresh and last run's frames stop being kept alive.
9245 reset_fingerprint_memo()
9246 # BUG-103: every stored dataset's tables, open or not — the Data page counts
9247 # them all — are held run after run and never written into, so each is
9248 # hashed once rather than on every rerun, however it entered the store
9249 # (the wizard, ✏️ Edit dataset, the recovery cache, a repair).
9250 for stored in (st.session_state.get("_datasets") or {}).values():
9251 if isinstance(stored, dict):
9252 vouch_for_frames(
9253 tuple(stored.get(t) for t in ("words", "fixations", "raw_gaze"))
9254 )
9255 st.session_state[_PLACEHOLDER_SHOWN_KEY] = False
9256 # BUG-96: the missing-corpus note describes the run that wrote it. It is
9257 # consumed later in that run, but a run that leaves before then — a mapping
9258 # the demo stand-in can't satisfy, a stopped or abandoned run — used to hand
9259 # it to the next run, which showed it over another dataset and hid the
9260 # dataset card's row counts.
9261 st.session_state.pop(_UNAVAILABLE_KEY, None)
9262 # Start capturing log records into the in-app debug buffer before any data
9263 # or plot work runs, so the debug panel (?debug=1) sees this run's logs.
9264 install_log_capture()
9265 # #374 F9: before any widget, so a write a closed popover would undo holds.
9266 reassert_pending_writes()
9267 announce_start_over()
9268 # Apply deep-link presets BEFORE any widget renders — see _apply_url_preset
9269 # for the full URL schema. External tools can deep-link into this app with
9270 # `?source=...&participant=...&trial=...&...` to land on a specific trial
9271 # with the reviewer's preferred viz settings.
9272 keys_before_link = set(st.session_state.keys())
9273 url_source = _apply_url_preset()
9274 # #374 F14: what the link seeded, to take back if it names a dataset that
9275 # isn't here (`resolve_link_dataset`, after the recovery cache restores).
9276 link_seeded = set(st.session_state.keys()) - keys_before_link
9277 links_dataset = bool(st.query_params.get(PARAM_DATASET))
9278 # ENG-26: desktop/localhost installs remember uploaded datasets, annotations,
9279 # mappings and view settings across browser refreshes and process restarts.
9280 # Public deployments never opt in implicitly (there is no user identity with
9281 # which to isolate the cache). URL presets are applied first and therefore
9282 # win over restored settings; an explicit ?source= also wins over the stored
9283 # data-source choice.
9284 app_url = str(getattr(st.context, "url", "") or "")
9285 # UX-166: restoring large uploads reads their Parquet files before even the
9286 # title is drawn, so it gets a card of its own at the very top of the page —
9287 # only while there is something to restore: once the session has had its
9288 # one attempt the call returns at once, and a card around it would cost
9289 # every run a timer thread and a task for nothing. Its slot is held on every
9290 # run either way (UX-167: one element fewer here would shift the view).
9291 restore_slot = st.empty()
9292 restoring = (
9293 contextlib.nullcontext()
9294 if local_state_restored(st.session_state)
9295 else loading.card(
9296 restore_slot,
9297 key="restore",
9298 title="Restoring your datasets from this computer",
9299 )
9300 )
9301 with restoring:
9302 restored = restore_local_state(
9303 st.session_state, app_url, protect_data_source=url_source is not None
9304 )
9305 # UX-167: Streamlit matches a rerun's elements to the last run's by position,
9306 # so a notice that shows on one run only (the unavailable-link warning below)
9307 # must not move the page under it — it writes into this container, drawn on
9308 # every run. The toasts need no slot: `st.toast` draws in Streamlit's event
9309 # container, outside the page.
9310 page_notices = st.container()
9311 if restored:
9312 # ENG-30: say it once, where the user is looking. Silently repopulating a
9313 # session reads as "the app kept my data somewhere" without saying where;
9314 # the toast points at the menu panel that answers that.
9315 #
9316 # UX-136: `restore_local_state` is true only when something the user
9317 # would *recognise* came back — a dataset, an annotation, a saved design.
9318 # Every rerun writes the cache, so a session that only ever changed view
9319 # settings still leaves a manifest, and announcing that as "your last
9320 # session" was wrong in the one case where it is most alarming: right
9321 # after the user cleared the cache by hand. Naming the counts is what
9322 # makes the claim checkable against the panel it points at.
9323 st.toast(
9324 f"Recovered {_restored_recap()} from this computer — see {ICONS['view_data']} Data Management → "
9325 "Saved on this computer.",
9326 icon=ICONS["recovery"],
9327 )
9328 elif consume_restore_skipped(st.session_state):
9329 # BUG-71: the last launch that restored the cache never finished, so this
9330 # one opened without it rather than failing the same way again.
9331 st.toast(
9332 "Your last session didn't finish opening, so it wasn't restored this "
9333 "time. It is still saved on this computer and saving is paused, so it "
9334 "stays that way: reload to try again, or delete it with "
9335 "`scanpath-studio cache --clear`.",
9336 icon=ICONS["warning"],
9337 duration="long",
9338 )
9339 # A cache that is there but did not all come back says so on every run
9340 # until it is retried or removed — never silently, and never by replacing
9341 # it (the held-back parts are kept by every save).
9342 render_cache_recovery_notice(page_notices, app_url, key="cache_recovery")
9343 linked_choice = None
9344 if url_source == "onestop" and onestop_data_dir() is not None:
9345 linked_choice = st.session_state.setdefault(
9346 "data_source_choice", ONESTOP_CHOICE
9347 )
9348 elif url_source == "multipleye" and multipleye_bundle_dir() is not None:
9349 linked_choice = st.session_state.setdefault(
9350 "data_source_choice", MULTIPLEYE_BUNDLE_CHOICE
9351 )
9352 elif url_source == "demo":
9353 linked_choice = st.session_state.setdefault("data_source_choice", DEMO_CHOICE)
9354 elif url_source == "synthetic":
9355 linked_choice = st.session_state.setdefault(
9356 "data_source_choice", SYNTHETIC_CHOICE
9357 )
9358 elif url_source == "author":
9359 linked_choice = st.session_state.setdefault("data_source_choice", AUTHOR_CHOICE)
9360 elif (
9361 url_source in ONESTOP_REGIME_TOKEN_CHOICES
9362 or url_source == ONESTOP_LEGACY_SOURCE_TOKEN
9363 ) and public_datasets_enabled():
9364 # DATA-3: the public OneStop corpus is shareable. DATA-63: one dataset
9365 # per regime, each its own token; a DATA-3 link's `onestop_public` names
9366 # its regime in `onestop_regime` (seeded by _apply_url_preset).
9367 regime = st.session_state.get("onestop_regime")
9368 linked_choice = st.session_state.setdefault(
9369 "data_source_choice",
9370 ONESTOP_REGIME_TOKEN_CHOICES.get(url_source)
9371 or ONESTOP_REGIME_CHOICES.get(regime, ONESTOP_REGIME_CHOICES["ordinary"]),
9372 )
9373 elif url_source == CORPUS_SOURCE_TOKEN:
9374 # DATA-27 (Task 12): `?source=corpus&corpus=<slug>` names ONE entry of
9375 # `public_dataset_registry()` — a built-in public corpus or a locally
9376 # prepared one, identically. Writing the registry label into
9377 # `data_source_choice` is all the picker's healing path needs: it accepts
9378 # the label as an entry and collapses it back to PUBLIC_DATASETS_CHOICE,
9379 # re-stashing it on `public_dataset_choice` itself. Seeding that key here
9380 # as well would be a second copy of the same answer — and a `setdefault`
9381 # of it is dead anyway, since a recovery-cached value (the only case
9382 # where seeding could matter) is exactly what `setdefault` won't replace.
9383 #
9384 # The resolution is gated on the feature flag, but the *message* is not:
9385 # a build with public datasets switched off can't open the corpus either,
9386 # and saying nothing at all would leave the recipient with a link that
9387 # silently did nothing.
9388 slug = str(st.query_params.get(PARAM_CORPUS) or "")
9389 corpus_choice = (
9390 corpus_choice_for_slug(slug) if public_datasets_enabled() else None
9391 )
9392 if corpus_choice:
9393 linked_choice = st.session_state.setdefault(
9394 "data_source_choice", corpus_choice
9395 )
9396 elif slug:
9397 # The common case, not an edge case: the recipient has no prepared
9398 # bundle, or a different subset of one. Say which corpus was named
9399 # and leave the picker exactly where it was — never wedge it, and
9400 # never silently open a different corpus. There is no remedy to name
9401 # (DATA-55): the app no longer discovers corpora, and until DATA-56's
9402 # add-from-a-folder flow nothing in it adds one.
9403 page_notices.warning(
9404 f"This link opens the dataset `{slug}`, which isn't available "
9405 "here. The link's view settings still apply to whatever you open."
9406 )
9407 elif url_source == "upload":
9408 st.session_state.setdefault("_show_upload_wizard", True)
9409 elif links_dataset:
9410 # #374 F14: `?dataset=` names a dataset the sender added. Opened when
9411 # this session holds one of that name — once, on the run that read the
9412 # link, so the picker stays the user's afterwards — and otherwise the
9413 # link is set aside whole, with a notice naming what is missing.
9414 current = st.session_state.get("data_source_choice")
9415 if (name := resolve_link_dataset(link_seeded, current)) is not None:
9416 linked_choice = name
9417 if st.session_state.get(_LINK_DATASET_OPENED_KEY) != name:
9418 st.session_state[_LINK_DATASET_OPENED_KEY] = name
9419 st.session_state["_pending_source_choice"] = name
9420 if notice := link_dataset_notice(st.session_state.get("data_source_choice")):
9421 page_notices.warning(notice, icon=ICONS["warning"])
9422 # EXP-19: the canvas / font a link seeded belong to the source it names.
9423 # Scope their protection from the source snap to that source — or drop it
9424 # when the link named none this app can open, so the fallback source still
9425 # snaps to its own monitor rather than wearing another corpus' canvas.
9426 scope_link_setup(linked_choice)
9428 # Chrome first, page heading second: Streamlit's native top nav, then the
9429 # settings menu bar, then the title.
9430 #
9431 # AFTER `restore_local_state`, deliberately: `render_nav` resolves the active
9432 # view and writes `main_nav`, so running it earlier would pin the view to the
9433 # router's default before the recovery cache could restore the one the user
9434 # was last on. BEFORE any data loading, equally deliberately: the loaders'
9435 # directory inputs, download buttons and column-mapping panels fill the bar's
9436 # popovers by `host=`, so those slots have to exist first — the same
9437 # reserve-then-fill discipline the former sidebar containers had.
9438 seed_debug_mode() # UX-37: a legacy ?debug=1 link pre-arms the Help toggle.
9439 # UX-62: before the nav — `st.logo` writes into the same header strip
9440 # `st.navigation` draws into, and has to be there when it renders.
9441 render_app_logo()
9442 # Active top-level view, resolved BEFORE `render_top_menu` so the BUG-31
9443 # wizard hold-override below can land before that call rather than after it.
9444 active_view = render_nav()
9446 # BUG-31 — navigating away mid-wizard used to land the user on the *half-built*
9447 # dataset: `resolve_data_source` reports `UPLOAD_CHOICE` for the whole
9448 # run while the wizard is open, whatever the view, so `main` returned early
9449 # with "this dataset isn't set up yet" over a session that still held every
9450 # finished dataset. It read as data loss; it was an unfinished wizard.
9451 #
9452 # While the wizard is open the 🗂️ Data page is what renders, wherever the nav
9453 # says the user is — and the wizard asks whether to discard. It has to keep
9454 # **rendering** for the question to be worth asking: Streamlit drops a
9455 # widget's key at the end of any run in which it did not render, and
9456 # `st.file_uploader` is the one widget `persist_state="session"` cannot cover
9457 # (ENG-36), so a single run spent drawing Scanpath throws the uploaded files
9458 # away before the user can be asked about them.
9459 #
9460 # Deliberately **no navigation of our own** — not a `switch_to_view`, not a
9461 # `main_nav` write. Both end in `st.switch_page`, which aborts the run where
9462 # it is called, and an aborted run renders no wizard: the fix would destroy
9463 # exactly what it exists to protect. So the nav highlight is simply allowed to
9464 # sit on the view the user picked while the page under it asks the question,
9465 # and *Discard* then needs no navigation at all — the router is already there.
9466 #
9467 # BUG-36 follow-up: resolved here, before `render_top_menu`, and handed in
9468 # as `active_view=`, so everything drawn after it agrees on the view.
9469 if st.session_state.get("_show_upload_wizard"):
9470 if (
9471 active_view != _VIEW_DATA
9472 and st.session_state.get(WIZARD_STAY_KEY) != active_view
9473 ):
9474 st.session_state[WIZARD_LEAVE_KEY] = active_view
9475 else:
9476 st.session_state.pop(WIZARD_LEAVE_KEY, None)
9477 active_view = _VIEW_DATA
9479 menu = render_top_menu(active_view=active_view)
9480 _render_about_panel(menu.title)
9481 _render_cancelled_load_notice(menu.notices)
9482 _render_backup_reminder(menu.notices, active_view)
9484 def _finish_page() -> None:
9485 """The run's last UI, once, on whichever path ``main`` leaves by.
9487 🗂️ Data → *Saved on this computer* reports what **this run** just
9488 persisted, so it has to come after `save_local_state` — which the early
9489 returns never reach. Each of them calls this instead, and the epilogue
9490 calls it for the ordinary path, so exactly one runs per script run.
9491 That is what keeps its widgets single.
9493 The section is drawn only while the overview is on screen: not on the
9494 other views, not under ✏️ Edit dataset, and not while the add-dataset
9495 wizard owns the page.
9496 """
9497 if data_view and not editing and not wizard_owns_page:
9498 _render_download_folder_section(download_folder_slot)
9499 _render_saved_here_section(app_url, saved_here_slot)
9501 # First-visit welcome tour. After the URL presets, so embeds and
9502 # deep-linked sessions can suppress it — but BEFORE the heavy data/plot
9503 # work, so the welcome streams to the browser immediately instead of
9504 # after the full first render. Replay clicks arm the tour in the button's
9505 # on_click callback, which runs before this point in the rerun.
9506 #
9507 # UX-167: one container for the whole block, so it always holds exactly one
9508 # index below it — `render_easter_egg()` is the case that bit: it draws a
9509 # bare iframe except while a tour or tutorial is active, so the first run
9510 # after one closes used to insert an element above the view and shift it.
9511 # See scanpath_studio/CLAUDE.md → Gotchas.
9512 with st.container():
9513 maybe_show_welcome_tour()
9514 render_spotlight_tour()
9515 # UX-40: task-oriented tutorials share the spotlight mechanism but keep
9516 # their own progress and do not inherit the welcome tour's opt-out.
9517 render_use_case_tutorial()
9518 # UX-39: arm the title's easter egg. After the tour/tutorial renders, because
9519 # its suppression reads the `tour_mode` / `tutorial_active` those set — and it
9520 # doesn't care about DOM order, being a height-0 script that retries until the
9521 # heading has hydrated.
9522 render_easter_egg()
9523 # UX-196: a cut-off dropdown label shows in full on hover. Unconditional,
9524 # so it never moves the view's index (UX-167).
9525 render_truncation_tooltips()
9526 # UX-15: same deal for the FAQ dialog — the ❓ Help menu button that arms it
9527 # renders at the bottom of this function, so serving it here is what keeps
9528 # the modal from waiting out the whole rerun. Ditto ℹ️ About, a dialog since
9529 # the menu bar made it a popover inside a popover.
9530 maybe_show_faq()
9531 maybe_show_about()
9532 maybe_show_tutorial_library()
9533 # UX-179 — ❓ Help → About → Debug, served early like its siblings.
9534 maybe_show_debug()
9536 # `active_view` was already resolved above (including the BUG-31 wizard
9537 # hold-override — see the note there); the dispatch below reuses it.
9539 # UX-166 — the Scanpath and Corpus views render inside one reserved area.
9540 # Its first child is the page slot: `loading.Page` fills it with a skeleton
9541 # of the view and the dataset card while a load is slow, and CSS hides the
9542 # rest of the area meanwhile — the previous page, or the new one being laid
9543 # out under it. It replaces the old "Loading <view>…" bridge, which a view
9544 # switch now shows at once as that view's skeleton. The slot is recreated
9545 # empty on every run, so nothing it held can outlive the run (BUG-81).
9546 view_area = st.container(key=loading.VIEW_AREA_KEY)
9547 view_first_slot = view_area.empty()
9548 # UX-167: reserved right after the page slot — creation order is screen
9549 # order, so this is always the view area's second child — and given to the
9550 # conditional notices below (`view_notices`) so the view's own blocks after
9551 # them keep their place whether or not a notice draws this run.
9552 view_notices_slot = view_area.container()
9553 if active_view == _VIEW_DATA:
9554 # UX-166: the Data page draws outside the area, so on that view it only
9555 # ever holds what the previous view left there — until this run ends,
9556 # the old page, above the new one. This marker hides it (styles.py) for
9557 # the whole run; it is not the page card's, which the Data page outlives.
9558 view_first_slot.markdown(
9559 '<span class="sps-view-hidden" aria-hidden="true"></span>',
9560 unsafe_allow_html=True,
9561 )
9562 view_switched = st.session_state.get("_last_rendered_view") not in (
9563 None,
9564 active_view,
9565 )
9566 st.session_state["_last_rendered_view"] = active_view
9568 # DATA-26 — the **Data** page ("Data Management"). One place for
9569 # everything about the dataset itself, instead of a ⚙️ Configure menu group
9570 # at the top and a 🔎 Data Inspection subtab on the far side of the Scanpath
9571 # view asking the same questions at two points in the pipeline.
9572 #
9573 # The page is built on EVERY run and merely hidden when another view is
9574 # active (`DATA_PAGE_OFFSCREEN_KEY` → `display: none` in styles.py). That is
9575 # not laziness: the slots below are filled by the loaders' directory input,
9576 # ⬇ Download button, source options and column-mapping selectboxes, all of
9577 # which *drive* `prepare_data` on every rerun — and Streamlit drops the key
9578 # of a widget that did not render. Rendering-then-hiding keeps the popovers'
9579 # every-run semantics exactly, so this stays a re-host rather than a rewrite
9580 # of every loader into a render/resolve pair.
9581 #
9582 # DATA-35 split it into **two screens**, both built every run and switched by
9583 # key for the same reason the page itself is (above). UX-197 made them two
9584 # *parts* of one page: the overview always shows, and the editor opens
9585 # under it:
9586 #
9587 # Overview 📂 Available datasets (the table + ➕ Add dataset)
9588 # 🔎 What's in the open dataset
9589 # Editor ✏️ Edit dataset — everything that *configures* the dataset:
9590 # description / options / data location, column mapping,
9591 # recording setup, trial identity, stimulus images, the two
9592 # metadata tables, preprocessing.
9593 #
9594 # The overview used to carry all of it in one scroll, which put a forty-row
9595 # table, twenty mapping selectboxes and three uploaders on the screen you
9596 # visit to answer "which datasets do I have?". Sub-slots are reserved in
9597 # page order and filled at whatever point of the load reaches them
9598 # (Streamlit lays containers out in creation order).
9599 data_view = active_view == _VIEW_DATA
9600 setup_page = st.container(
9601 key=DATA_PAGE_KEY if data_view else DATA_PAGE_OFFSCREEN_KEY
9602 )
9603 # UX-66: while the add-dataset wizard is up it owns the page and carries its
9604 # own sticky title, so the page's header and its stage subheading would be a
9605 # second and third title above it.
9606 wizard_owns_page = bool(st.session_state.get("_show_upload_wizard"))
9607 editing = (
9608 bool(st.session_state.get(DATASET_EDITOR_OPEN_KEY)) and not wizard_owns_page
9609 )
9610 # UX-197: the overview stays on screen while the editor is open, and the
9611 # editor opens under it, set apart — beta testers lost the dataset's counts
9612 # and tables the moment they started editing it. Only the editor is
9613 # switched by key now.
9614 overview_page = setup_page.container(key=DATA_OVERVIEW_KEY)
9615 editor_page = setup_page.container(
9616 key=DATA_EDITOR_KEY if editing else DATA_EDITOR_OFFSCREEN_KEY
9617 )
9618 # UX-177 — the page title, then the table of datasets straight under it:
9619 # no "Available datasets" subheading and no rule between the two, since the
9620 # page is about nothing else until *What's in the dataset* below.
9621 if data_view and not wizard_owns_page:
9622 overview_page.header("Data Management")
9623 setup_source_slot = overview_page.container()
9624 # The editor's own header bar — the ✏️ Edit dataset screen's title and its
9625 # way back, filled below once the dataset's display name is known.
9626 editor_head_slot = editor_page.container()
9627 # UX-166: the dataset card's slot on the ✏️ Edit dataset screen, directly
9628 # under its header bar: opening the editor scrolls the page down to it
9629 # (UX-197), so that is where the user is looking.
9630 editor_loading_slot = editor_page.empty()
9631 # UX-135 — the editor's sections are the add screen's numbered *parts*, not
9632 # a `st.divider()` + `st.subheader()` + `st.caption()` stack. `_editor_part`
9633 # below draws one headline into each of the slots reserved here; the
9634 # registry, the numbering and the hover note are `wizard_shell.EDITOR_STEPS`.
9635 #
9636 # Part 2 (Data tables & column mapping) opens above the public loader's
9637 # captions because everything down to the metadata tables belongs to it —
9638 # where the files are, how their columns map, and what is attached to them
9639 # — exactly as the add screen's part 2 holds every upload and mapping row.
9640 # UX-178 — part 1 is the add screen's: Name & description.
9641 editor_part_name_slot = editor_page.container()
9642 editor_part_data_slot = editor_page.container()
9643 description_slot = editor_page.container()
9644 source_options_slot = editor_page.container()
9645 data_location_slot = editor_page.container()
9646 # UX-106 — "add the AOI/fixations table this dataset was created without"
9647 # is an *upload*, and the add screen puts every upload above the mapping.
9648 # Reserved here so it renders there; filled from `_render_remap_editor`,
9649 # which runs much later (creation order is screen order, so the two need
9650 # not agree).
9651 editor_uploads_slot = editor_page.container()
9652 # The add-dataset wizard takes the whole page, so its slot is the page's,
9653 # not either screen's.
9654 setup_wizard_slot = setup_page.container()
9655 # UX-52 round 2 — "what's in this dataset" comes **before** the mapping, on
9656 # the user's call. It breaks pipeline order deliberately: the counts are the
9657 # first thing you want after choosing a source ("did it load, and is it the
9658 # right size?"), and the mapping is what you scroll to when the answer looks
9659 # wrong. DATA-35 kept that order across the split: the counts are the
9660 # overview's second half, and the mapping opens the editor.
9661 setup_body_slot = overview_page.container()
9662 # UX-179 — *Saved on this computer*, the overview's last section: the
9663 # recovery cache and the two ways to throw work away. Filled by
9664 # `_finish_page`, after this run's `save_local_state`.
9665 download_folder_slot = overview_page.container(key="data_download_folder")
9666 saved_here_slot = overview_page.container(key="data_saved_here")
9667 # Keyed → the stable `.st-key-…` selectors the "Load and verify a dataset"
9668 # tutorial spotlights (UX-40), alongside `tutorial_data_inspection` above.
9669 column_mapping_slot = editor_page.container(key="tutorial_column_mapping")
9670 mapping_body_slot = column_mapping_slot.container()
9671 # Raw tables for a dataset whose mapping is still broken. Its own slot,
9672 # *below* the mapping editor, which is the control that fixes them.
9673 unmapped_slot = editor_page.container()
9674 # DATA-20 §1 — the participant/trial/text metadata tables. After the mapping
9675 # (they join on the reader id the mapping just settled), and — UX-135 —
9676 # *inside* part 1 with it, under their own small **Metadata** heading, which
9677 # is exactly where the add screen puts the same three uploaders.
9678 setup_metadata_slot = editor_page.container(key="tutorial_participant_metadata")
9679 # UX-135 — Recording setup is the add screen's part 3, so it is the editor's
9680 # own numbered part too rather than a `##### ` heading buried at the end of
9681 # the mapping form. Filled from `tabs._render_column_mapping_section`, which
9682 # is handed this slot: it is the same renderer either way, only re-hosted.
9683 setup_recording_slot = editor_page.container(key="tutorial_recording_setup")
9684 # UX-52 round 3 — the VAL-7 trial-identity verdict is its own section, not a
9685 # `#####` item inside "What's in this dataset" (the user's call). It carries
9686 # a *verdict* — sometimes a warning — and the fix it names is a change to the
9687 # Trial ID mapping directly above it, so it belongs at the same level as the
9688 # thing it judges rather than buried under the counts.
9689 # Keyed → the `.st-key-…` selector the "Load and verify a dataset" tutorial
9690 # spotlights, alongside its siblings above and below.
9691 setup_identity_slot = editor_page.container(key="tutorial_trial_identity")
9692 # VIZ-14: local stimulus-image paths, after the questions that describe the
9693 # data itself.
9694 setup_stimulus_slot = editor_page.container(key="tutorial_stimulus_images")
9695 setup_preproc_slot = editor_page.container(key="tutorial_preprocessing")
9696 # UX-106 — the editor's own foot: ✅ Save changes, under everything it
9697 # saves, the way ✅ Add dataset sits under the whole add screen. Reserved
9698 # last so it lands after preprocessing; filled at the end of the run.
9699 editor_footer_slot = editor_page.container(key="dataset_editor_footer")
9701 # UX-135 — which of the editor's five parts are on screen this run, and so
9702 # what each one is numbered. Two are conditional (stimulus images need a
9703 # local filesystem; preprocessing is behind PRE-22's flag), and a screen
9704 # reading 1 · 2 · 3 · 5 looks like a section that failed to render rather
9705 # than one that does not apply here.
9706 _editor_shown = {"edit_name", "edit_data", "edit_setup", "edit_identity"}
9707 if local_filesystem_enabled():
9708 _editor_shown.add("edit_stimulus")
9709 if preprocessing_enabled():
9710 _editor_shown.add("edit_preproc")
9711 editor_parts = wizard_shell.numbered(wizard_shell.EDITOR_STEPS, _editor_shown)
9713 def _editor_part(host, step_id: str):
9714 """One numbered editor part's headline; returns the body to fill.
9716 The add screen's `wizard_shell.part`, verbatim — same chip, same rule
9717 above it, same hover note — so the two screens read as one screen before
9718 and after the dataset exists.
9719 """
9720 step = editor_parts[step_id]
9721 return wizard_shell.part(host, step, note=step.caption)
9723 # Data source selection. UX-25: only the *resolution* happens here (it must
9724 # precede the load); the picker itself renders in the main view — on the
9725 # Scanpath "Filter by" row, at the top of the Corpus view, or (DATA-26) in
9726 # the page slot above. The resolver takes the same slot because while the
9727 # add-dataset wizard is open it renders a "✕ Cancel" bar *instead of* a
9728 # picker, and rendering both would duplicate the `tour_grp_data_source` key.
9729 from scanpath_studio.wizard import _enter_add_data_wizard
9731 data_choice = resolve_data_source(host=setup_source_slot)
9732 # DATA-47 — the metadata tables belong to a dataset. Swap the selected one's
9733 # onto the session keys every consumer reads (`metadata.active()` & co.),
9734 # filing the previous dataset's away. Keyed by the concrete canonical choice
9735 # the dataset table uses, not `data_choice` — every public corpus loads
9736 # through one category token, and they must not share a table. The add
9737 # wizard's dataset has no name yet, so it gets the pending slot.
9738 # DATA-48 — and so do the annotations, swapped by the same key.
9739 import scanpath_studio.annotations as _annotations
9740 from scanpath_studio import metadata as _metadata
9742 _dataset_owner = (
9743 _metadata.PENDING_DATASET
9744 if data_choice == UPLOAD_CHOICE
9745 else str(st.session_state.get("data_source_choice") or data_choice)
9746 )
9747 # A cancelled edit's metadata tables go back first, to the dataset they
9748 # were taken from, before the swap below files them away.
9749 apply_editor_restore()
9750 _metadata.activate_dataset(st.session_state, _dataset_owner)
9751 # Adoption of an old cache's unassigned entries waits for the load, which
9752 # says what is really shown (`_file_annotations_under_shown_dataset`).
9753 _annotations.activate_dataset(
9754 st.session_state, annotations_owner(_dataset_owner), adopt=False
9755 )
9756 # UX-166: on the Data page the dataset card sits above the table.
9757 data_page_slot = setup_source_slot.empty()
9758 # UX-54: the page lists every dataset as a *table* — one row each, sortable,
9759 # with the counts beside the name and the per-row actions in the row they
9760 # belong to. Reserved here (so it keeps its place at the top of the page)
9761 # and filled after the load, which is the first point this run's counts for
9762 # the open dataset exist.
9763 # Keyed: the "Load and verify a dataset" tutorial spotlights it — the data
9764 # source picker it used to aim at is not on this page (only Scanpath and
9765 # Corpus Analysis draw one), so the step outlined nothing.
9766 dataset_table_slot = setup_source_slot.container(key="tutorial_available_datasets")
9767 # UX-174 r2 → UX-177 — the two ways to make a dataset are one **+ Add
9768 # dataset** menu (the Scanpath picker's + with its name spelled out), under
9769 # the list it adds to. Filled once `data_choice` is known.
9770 add_dataset_slot = (
9771 setup_source_slot.container(key="data_page_add")
9772 if data_view and not wizard_owns_page
9773 else None
9774 )
9775 # UX-166: this run's page — the slot the skeleton and the dataset card draw
9776 # into while a load is slow: the view area's first child on Scanpath and
9777 # Corpus Analysis; on the Data page, the slot above the table, or the one
9778 # under the ✏️ Edit dataset header while the editor is open.
9779 if not data_view:
9780 page_slot = view_first_slot
9781 elif editing:
9782 page_slot = editor_loading_slot
9783 else:
9784 page_slot = data_page_slot
9785 page = loading.page(
9786 page_slot,
9787 view="data"
9788 if data_view
9789 else "corpus"
9790 if active_view == _VIEW_CORPUS
9791 else "scanpath",
9792 plot_height=loading.recorded_plot_height("single", 480),
9793 )
9794 if data_view and add_dataset_slot is not None and data_choice != UPLOAD_CHOICE:
9795 # UX-64 took ➕ Add data off the Scanpath row and made this page the only
9796 # way in — so the way in has to *be* here. Without this menu
9797 # `_enter_add_data_wizard` would have no trigger at all and uploading
9798 # would be unreachable. `on_click` callbacks, not inline handlers: they
9799 # reassign `data_source_choice`, which only lands before the widgets
9800 # instantiate.
9801 with add_dataset_slot.popover(
9802 "Add dataset", icon=ICONS["add"], type="primary", key="data_add_menu"
9803 ):
9804 st.button(
9805 "Create manually",
9806 icon=ICONS["author"],
9807 key="create_manual_scanpath_btn",
9808 on_click=_enter_manual_dataset,
9809 help="Write a text and place its fixations by hand. Saved "
9810 "scanpaths join the list of datasets.",
9811 width="stretch",
9812 )
9813 st.button(
9814 "Import files",
9815 icon=ICONS["upload"],
9816 key="add_data_btn",
9817 on_click=_enter_add_data_wizard,
9818 help="Add your Fixations and Words (interest areas) tables.",
9819 width="stretch",
9820 )
9821 # UX-178 — part 1's headline on every run, like the other parts (the editor
9822 # is built hidden); its two fields only while it is open, since they are
9823 # seeded from whichever dataset is open and must not outlive it.
9824 editor_name_body = (
9825 _editor_part(editor_part_name_slot, "edit_name") if data_view else None
9826 )
9827 if data_view and editing:
9828 _render_dataset_editor_bar(editor_head_slot, data_choice)
9829 # UX-174 r2 — the description is edited here, with the rest of the
9830 # dataset, at the top of part 1 (the add screen asks for it beside the
9831 # name). The public loader's own caption lands under it, in this slot.
9832 editing_token = str(st.session_state.get("data_source_choice") or data_choice)
9833 hold_editor_staging(editing_token)
9834 render_name_field(editor_name_body, editing_token)
9835 render_description_field(editor_name_body, editing_token)
9836 # PRE-22: the section is held back from this release — heading, caption and
9837 # controls all come from behind the same gate, so the page has no gap where
9838 # a hidden stage used to be.
9839 if data_view and preprocessing_enabled():
9840 _editor_part(setup_preproc_slot, "edit_preproc")
9842 preproc_settings = _activate_data_source(
9843 data_choice, preproc_host=setup_preproc_slot
9844 )
9845 # UX-166 — one card over the dataset pipeline, drawn in the page slot. The
9846 # post-wizard "Dataset added — loading your scanpaths…" bridge it replaces
9847 # is this card shown at once.
9848 dataset_card = _open_dataset_card(
9849 page,
9850 data_choice,
9851 view=active_view,
9852 view_switched=view_switched,
9853 finalizing=bool(st.session_state.pop("_wizard_finalizing", False)),
9854 )
9856 def _end_loading(*, showing_dataset: bool = True) -> None:
9857 """Take the dataset card and the page skeleton down on an early return.
9859 BUG-81: only the normal path cleared the old loading banners, so every
9860 early return (the wizard, a mapping that can't be satisfied, a filter
9861 that empties the pool) left one above the real content until the next
9862 click — on the very page the warning had just sent the user to.
9864 ``showing_dataset``: the run leaves the dataset on screen — a mapping
9865 still to fix, a filter that emptied the pool — so it is where Cancel
9866 goes back to (UX-168); the add-dataset wizard's return shows none.
9867 """
9868 if dataset_card is not None:
9869 dataset_card.close()
9870 st.session_state.pop(DATASET_TASK_KEY, None)
9871 page.release()
9872 if showing_dataset:
9873 _remember_open_dataset(data_choice)
9875 def _render_datasets_table(
9876 words, fixations, raw_gaze, *, scope_note: str | None = None
9877 ) -> None:
9878 """📂 Available datasets, whenever the Data page is showing its overview.
9880 BUG-81: this used to render only after a successful load, so a dataset
9881 whose mapping can't be satisfied — or a filter that empties the pool —
9882 left the heading with no table under it, and no way to switch to
9883 another dataset short of ♻️ Reset.
9884 """
9885 if not data_view or wizard_owns_page:
9886 return
9887 # UX-107 — ✅ Save changes closes the editor, so its success line
9888 # belongs here, on the screen it returns to.
9889 saved = st.session_state.pop("_remap_applied", None)
9890 join_notices = st.session_state.pop(STIMULUS_JOIN_NOTICE_KEY, None)
9891 builtin_saved = st.session_state.pop(BUILTIN_MAPPING_SAVED_KEY, None)
9892 if builtin_saved is not None:
9893 dataset_table_slot.success(
9894 f"**{_dataset_display_name(str(builtin_saved))}** updated — "
9895 "your changes are saved.",
9896 icon=ICONS["success"],
9897 )
9898 if saved:
9899 dataset_table_slot.success(
9900 f"**{_dataset_display_name(str(saved))}** updated — mapping, "
9901 "recording setup and any table you added are saved.",
9902 icon=ICONS["success"],
9903 )
9904 # DATA-49: a save whose word boxes reached only some readings says
9905 # so here — its warning was raised inside the button's callback.
9906 for notice in join_notices or []:
9907 dataset_table_slot.warning(notice, icon=ICONS["warning"])
9908 # Rendered *inside* the slot rather than handed it: the table is a
9909 # fragment, and a fragment rerun may only draw widgets into its own
9910 # containers.
9911 with dataset_table_slot:
9912 render_dataset_table(
9913 # Public corpora load through the historical category token,
9914 # while the table rows use concrete registry labels. Preserve
9915 # that concrete canonical selection so the active row and its
9916 # remembered counts are keyed to the row the user can revisit.
9917 active=str(st.session_state.get("data_source_choice") or data_choice),
9918 words=words,
9919 fixations=fixations,
9920 raw_gaze=raw_gaze,
9921 scope_note=scope_note,
9922 )
9924 # (DATA-9's ordered source-config group — description · options · data
9925 # location · column mapping — is now the top of the Data page reserved
9926 # above. VIZ-31 had already moved "Experimental Setup" out of it: monitor
9927 # geometry, fonts, text colour and plot background are figure settings, and
9928 # they render in the Scanpath rail beside the layers they restyle.)
9930 # UX-166: what the pipeline draws on its way to the view belongs *above* it —
9931 # the ✏️ Author editor (the view's input), the data-quality warning and
9932 # UX-7(b)'s missing-corpus panel (notes on it). The Scanpath and Corpus
9933 # views render inside `view_area`, created before all of this, so on those
9934 # views these go into it too (after the page slot, so they wait under a
9935 # skeleton with the rest of the new page); the Data page keeps them where
9936 # they always were. UX-167: `view_notices_slot` is its own container, held
9937 # on every run, so the view's blocks below it keep their place whether or
9938 # not a notice draws this run.
9939 view_notices = contextlib.nullcontext() if data_view else view_notices_slot
9941 # Load + map core data. The **Upload** source renders each table as an
9942 # [upload box → mapping] group on the 🗂️ Data page (words, fixations, raw gaze) and
9943 # normalizes inline; every other source auto-detects (or, for public datasets,
9944 # renders standalone mapping panels) via prepare_data. Keep the raw frames
9945 # around so we can show them if the mapping isn't ready.
9946 #
9947 # Decide which participant (if any) the OneStop loader should fast-path to.
9948 # 1. A URL deep link (?participant=) → load just that pid's shard (embedded
9949 # review use case); captured once so the live selector can't change it.
9950 # 2. Otherwise, if the full CSV bundle exists → load the whole corpus once
9951 # (participant=None) and let in-app participant switching just *filter*
9952 # it — so changing participant is instant instead of re-invoking the
9953 # loader on every change.
9954 # 3. Shards-only setup with no full bundle → fall back to lazy per-pid
9955 # loading driven by the selector (the ~60 GB corpus can't be held whole).
9956 deeplink_pid = st.session_state.get("_deeplink_participant")
9957 if deeplink_pid:
9958 deep_link_pid = deeplink_pid
9959 elif data_choice == ONESTOP_CHOICE and not onestop_full_bundle_exists():
9960 deep_link_pid = st.session_state.get("single_participant")
9961 elif data_choice == MULTIPLEYE_BUNDLE_CHOICE:
9962 # MultiplEYE has no full-corpus bundle: each session is its own shard, so
9963 # the live participant selector fast-paths to one session's shards too.
9964 deep_link_pid = st.session_state.get("single_participant")
9965 else:
9966 deep_link_pid = None
9967 raw_gaze_df: pd.DataFrame | None = None
9968 # Did this load already draw the editable pre-normalization mapping panels
9969 # into the page's Column mapping section? (Mode A — see the dispatch below.)
9970 mapping_editor_rendered = False
9971 # Start each load with a clean column-mapping stash; each branch below
9972 # records the schema it used for the Data page's mapping section.
9973 _reset_active_mapping()
9974 if data_choice == UPLOAD_CHOICE:
9975 # Hybrid setup wizard: a guided flow on first load, then a compact
9976 # collapsed "Data & mapping" panel. DATA-26 made it the **Data page's
9977 # add-a-dataset mode** — adding a dataset *is* setup, and a wizard that
9978 # lived anywhere else would re-create the two-places-for-one-job problem
9979 # the page exists to fix. It still owns the page while active: there is
9980 # nothing for the other views to draw until it finishes, so `main`
9981 # returns here exactly as before. `_enter_add_data_wizard` requests the
9982 # Data view when the ➕ button is clicked, so the user is already here.
9983 wizard_active = not st.session_state.get("setup_complete", False)
9984 # Imported lazily (not at module top) to avoid the app⇄wizard import cycle.
9985 from scanpath_studio.wizard import _render_data_setup
9987 with setup_wizard_slot:
9988 setup = _render_data_setup(active=wizard_active)
9989 words_df, fixations_df = setup.words, setup.fixations
9990 raw_gaze_df = setup.raw_gaze
9991 raw_words_df, raw_fixations_df = setup.raw_words, setup.raw_fixations
9992 mapping_problems = setup.problems
9993 if wizard_active:
9994 _render_offpage_setup_notice(data_view)
9995 _finish_page()
9996 _end_loading(showing_dataset=False)
9997 return
9998 elif data_choice == MANUAL_SAMPLE_CHOICE and not _authoring_editor_open(
9999 data_choice
10000 ):
10001 # The example is shown like any stored dataset; ✏️ Edit opens its editor.
10002 words_df, fixations_df = _manual_sample_frames()
10003 raw_words_df, raw_fixations_df = words_df, fixations_df
10004 raw_gaze_df = pd.DataFrame()
10005 mapping_problems = []
10006 elif data_choice in (AUTHOR_CHOICE, MANUAL_SAMPLE_CHOICE):
10007 with view_notices:
10008 words_df, fixations_df = _render_authoring_source()
10009 if active_view == _VIEW_SCANPATH:
10010 # The authoring canvas is this screen's visualization.
10011 _finish_page()
10012 _end_loading()
10013 return
10014 raw_words_df, raw_fixations_df = words_df, fixations_df
10015 raw_gaze_df = pd.DataFrame()
10016 mapping_problems = []
10017 elif data_choice in st.session_state.get("_datasets", {}):
10018 # A dataset the user uploaded earlier and named — its frames were
10019 # normalized once by the wizard and stored in session, so switching back
10020 # to it is instant (no re-upload, no re-mapping). See _render_data_setup's
10021 # finalize and resolve_data_source.
10022 stored = st.session_state["_datasets"][data_choice]
10023 # DATA-39 — a dataset saved on ✏️ Edit dataset before that fix has its
10024 # AOI table stranded on the placeholder reader, so every scanpath drew
10025 # without its boxes and text. Repair it once, in the store itself, so
10026 # the recovery cache writes the repaired frames and it stays fixed.
10027 # A repair that cannot be made is not retried while the frames are the
10028 # same: the diagnosis is only "the flag is still set", so a failed
10029 # attempt would otherwise redo the whole harmonize on every rerun.
10030 failed = st.session_state.setdefault("_data39_repair_failed", {})
10031 attempt = frame_fingerprint(stored["words"])
10032 if failed.get(data_choice) != attempt:
10033 repaired = repair_stranded_stimulus_words(
10034 stored["words"], stored["fixations"]
10035 )
10036 if repaired is not None:
10037 stored = {**stored, "words": repaired[0], "fixations": repaired[1]}
10038 st.session_state["_datasets"][data_choice] = stored
10039 failed.pop(data_choice, None)
10040 elif STIMULUS_WORDS_FLAG in stored["words"].columns:
10041 failed[data_choice] = attempt
10042 words_df, fixations_df = stored["words"], stored["fixations"]
10043 raw_gaze_df = stored["raw_gaze"]
10044 # BUG-103: held run after run and never written into, so each is hashed
10045 # once — not on every rerun — even when it was read back from disk.
10046 vouch_for_frames((words_df, fixations_df, raw_gaze_df))
10047 raw_words_df, raw_fixations_df = words_df, fixations_df
10048 mapping_problems = []
10049 # Re-publish this dataset's chosen filter fields so the trial-filter
10050 # funnel offers the same dynamic conditions.
10051 st.session_state["wizard_filter_fields"] = list(stored.get("filter_fields", []))
10052 # Restore the composite trial-id components (session-only state) so the
10053 # trial picker renders its Participant/Text cascade — every other load
10054 # path sets this, but the stored branch doesn't re-normalize. Without it
10055 # the picker would inherit whatever source was loaded last.
10056 composite = list(stored.get("composite_trial_columns") or [])
10057 st.session_state["_composite_trial_columns"] = composite or None
10058 # Re-publish the stored column mapping so the Data Inspection tab shows
10059 # how this dataset's columns were mapped (the wizard isn't re-run here).
10060 # DATA-66: and its column-name map, which only the stored entry holds —
10061 # the raw tables it was built from are gone.
10062 stored_names = stored.get("column_names") or {}
10063 for table, schema in (stored.get("schemas") or {}).items():
10064 _stash_active_mapping(
10065 table,
10066 schema,
10067 names=ColumnNames.from_payload(stored_names[table])
10068 if table in stored_names
10069 else None,
10070 )
10071 else:
10072 # Built-in sources (demo / synthetic / OneStop / public) auto-detect
10073 # their mapping, so they skip the wizard entirely. Drop any wizard filter
10074 # fields left over from a prior upload so the funnel falls back to the
10075 # built-in default conditions for these sources.
10076 st.session_state.pop("wizard_filter_fields", None)
10077 # Re-propose the column mapping when the monitor-defining source changes.
10078 # The `col_map_*` widget keys persist across reruns, so a previous corpus'
10079 # mapping sticks to the next one — e.g. PoTeC maps Trial → `text_id`, and
10080 # since MultiplEYE *also* has a `text_id` column the stale-column reset
10081 # (which only fires when a mapped column vanishes) wouldn't catch it, so
10082 # MultiplEYE's per-page `trial_id` was ignored and every page collapsed
10083 # into one stimulus-level trial. Clearing on source change lets each
10084 # corpus auto-detect its own mapping; same-source reruns (and restores)
10085 # keep their keys. Mirrors the canvas re-seed in render_canvas_controls.
10086 #
10087 # Two harmonised benchmark corpora share one schema, so switching between
10088 # them re-proposes a mapping that auto-detects to the same thing — the
10089 # cost of one key covering every corpus, and the same trade every other
10090 # pair of sources already makes.
10091 source_key = (data_choice, st.session_state.get("public_dataset_choice"))
10092 if st.session_state.get("_colmap_seeded_for") != source_key:
10093 reset_column_mapping()
10094 st.session_state["_colmap_seeded_for"] = source_key
10095 restore_builtin_mapping(source_key)
10096 raw_words_df, raw_fixations_df = load_words_and_fixations(
10097 data_choice,
10098 participant=deep_link_pid,
10099 # DATA-9 ordered group: a public dataset renders its caption / source
10100 # options / data-location controls into these reserved sub-slots.
10101 description_host=description_slot,
10102 options_host=source_options_slot,
10103 location_host=data_location_slot,
10104 )
10105 # BUG-103: the loader's own ID for these tables, so the normalization
10106 # below is keyed on which load they came from rather than re-hashed.
10107 adopt_source(raw_words_df, raw_fixations_df)
10108 # DATA-48: the demo may have stood in for a corpus that isn't here.
10109 _file_annotations_under_shown_dataset(_dataset_owner)
10110 if dataset_card is not None and len(dataset_card.steps) == 3:
10111 if st.session_state.get(_UNAVAILABLE_KEY):
10112 # UX-166: the corpus isn't here, so the rows just read are the
10113 # bundled demo's stand-in — not counts for the corpus the card
10114 # names. The step moves on under its plain label.
10115 dataset_card.step(1)
10116 else:
10117 dataset_card.step(
10118 1,
10119 f"Mapping {len(raw_words_df):,} word rows and "
10120 f"{len(raw_fixations_df):,} fixations",
10121 )
10122 declared_word_schema, declared_fix_schema = declared_schemas_for(data_choice)
10123 mapping_editor_rendered = data_choice in (PUBLIC_DATASETS_CHOICE, DEMO_CHOICE)
10124 words_df, fixations_df, mapping_problems = prepare_data(
10125 raw_words_df,
10126 raw_fixations_df,
10127 # Show the Column-mapping panels for public datasets AND the Bundled
10128 # Demo (DATA-8) so the re-mapping capability is discoverable on the
10129 # default first-load source; pre-filled with auto-detection, so an
10130 # untouched mapping normalizes identically.
10131 allow_override=mapping_editor_rendered,
10132 # Mode A of the Data page's one Column mapping section (DATA-26).
10133 mapping_host=mapping_body_slot,
10134 # A prepared benchmark corpus publishes its schema; auto-detection
10135 # must not re-guess it from the publisher's leftover columns. The
10136 # panels stay editable — this only changes what they start at.
10137 declared_word_schema=declared_word_schema,
10138 declared_fix_schema=declared_fix_schema,
10139 # BUG-32: the add-dataset wizard writes these same `col_map_*` keys
10140 # and its field widgets persist, so coming back here from it would
10141 # otherwise inherit its picks whenever the headers match.
10142 mapping_dataset=source_key,
10143 # While ✏️ Edit dataset is open the panels are a draft and the
10144 # dataset keeps its mapping until ✅ Save changes.
10145 held_schemas=(
10146 held_builtin_mapping(source_key)
10147 if editing and mapping_editor_rendered
10148 else None
10149 ),
10150 )
10151 if mapping_editor_rendered:
10152 if editing:
10153 st.session_state[_REMAP_DIRTY_KEY] = builtin_mapping_is_dirty(
10154 source_key
10155 )
10156 else:
10157 hold_builtin_mapping(source_key)
10158 if not mapping_editor_rendered:
10159 # Another kind of source is open: no built-in snapshot may be restored
10160 # over the mapping keys it (or the add wizard) shares.
10161 st.session_state.pop(BUILTIN_MAPPING_HELD_KEY, None)
10162 st.session_state.pop(BUILTIN_MAPPING_RESTORE_KEY, None)
10163 if mapping_problems:
10164 # A required column is still unmapped. Rather than halt the whole app
10165 # (which hid the data the user needs to choose the mapping), show the
10166 # raw tables on the Data page, right under the still-editable Column
10167 # mapping section — and, from any other view, say where that page is.
10168 with unmapped_slot:
10169 _render_unmapped_view(raw_words_df, raw_fixations_df, mapping_problems)
10170 # BUG-100: the slot above is on the ✏️ Edit dataset screen, hidden until
10171 # it is opened — the overview needs its own word, where *What's in the
10172 # dataset* would have been — open editor or not, since UX-197 keeps
10173 # the overview on screen above it.
10174 if data_view and not wizard_owns_page:
10175 with setup_body_slot:
10176 _render_dataset_load_failure(
10177 _dataset_display_name(_dataset_owner), mapping_problems
10178 )
10179 _render_offpage_setup_notice(data_view)
10180 _finish_page()
10181 _render_datasets_table(None, None, None)
10182 _end_loading()
10183 return
10185 # VIZ-14: local/desktop users can attach stimulus screenshots without
10186 # adding an image_path column to their data. This intentionally stays out
10187 # of public deployments and share links because it contains machine-local
10188 # filesystem information; the same resolver is available through the API
10189 # and CLI for reproducible headless renders.
10190 if local_filesystem_enabled():
10191 with _editor_part(setup_stimulus_slot, "edit_stimulus"):
10192 image_root = st.text_input(
10193 "Image folder",
10194 key="stimulus_image_root",
10195 placeholder="/path/to/stimulus-images",
10196 help="Local folder containing one image per text or trial.",
10197 ).strip()
10198 image_pattern = st.text_input(
10199 "Filename pattern",
10200 key="stimulus_image_pattern",
10201 value="{text_id}.png",
10202 help="Use the app's field names in braces — {text_id}, {trial_id} "
10203 "or {participant_id}. Subfolders work too.",
10204 ).strip()
10205 if image_root:
10206 try:
10207 words_df = resolve_stimulus_image_paths(
10208 words_df, image_root, image_pattern
10209 )
10210 fixations_df = resolve_stimulus_image_paths(
10211 fixations_df, image_root, image_pattern
10212 )
10213 found = sum(
10214 _rows_with_local_images(frame)
10215 for frame in (words_df, fixations_df)
10216 )
10217 st.caption(f"Found a local image for {plural(int(found), 'row')}.")
10218 except ValueError as exc:
10219 st.error(f"Couldn't use this folder or pattern: {exc}")
10221 # Optional raw gaze: the Upload source already mapped + normalized it above;
10222 # every other source loads it here (bundled demo sample, OneStop uploader).
10223 if raw_gaze_df is None:
10224 raw_gaze_df = load_raw_gaze_data(
10225 data_choice, host=data_location_slot, notices=menu.notices
10226 )
10228 if preproc_settings["enabled"]:
10229 fixations_df, preproc_report = preprocess_fixation_stage(
10230 words_df, fixations_df, preproc_settings
10231 )
10232 st.session_state["_preprocessing_report"] = preproc_report
10233 st.session_state["_preprocessing_settings"] = dict(preproc_settings)
10234 suspicious = (
10235 preproc_report[
10236 preproc_report["suspicious_word_load"].fillna(False).astype(bool)
10237 ]
10238 if "suspicious_word_load" in preproc_report
10239 else preproc_report.iloc[0:0]
10240 )
10241 if not suspicious.empty:
10242 with view_notices:
10243 st.warning(
10244 f"Data quality: {plural(len(suspicious), 'trial')} put at least 12 "
10245 "fixations on one word. Check stimulus alignment or line "
10246 "assignment."
10247 )
10248 else:
10249 st.session_state["_preprocessing_report"] = pd.DataFrame()
10250 st.session_state["_preprocessing_settings"] = dict(preproc_settings)
10252 # UX-7(b): if the selected corpus isn't on disk, say so here — in the main
10253 # area, where the (demo) plot the user is actually looking at is — rather than
10254 # leaving it to a line on the 🗂️ Data page they may never open.
10255 with view_notices:
10256 _render_dataset_unavailable()
10258 # Whole-dataset frames, captured BEFORE the trial-filter funnel —
10259 # the Bulk Export tab's "Export the whole dataset" option exports these,
10260 # ignoring the current filters.
10261 words_all, fixations_all = words_df, fixations_df
10262 raw_gaze_all = raw_gaze_df
10263 # DATA-20: every reader in the dataset, before any narrowing — what the
10264 # participant-metadata join is reported against.
10265 #
10266 # Gated and cached, both deliberately. `participant_ids` is a `.unique()`
10267 # over *both unfiltered corpus frames* — ~0.5 s on full OneStop — and on
10268 # the default path (no table attached, not on the Data page) the answer is
10269 # thrown away, so an unconditional call put half a second on every rail
10270 # toggle and every ◀ ▶ step for nothing. The fingerprints are the ones
10271 # computed just below for the identity report, so the cache key is free.
10272 participants_all: list = []
10273 if data_view or metadata_mod.active() is not None:
10274 participants_all = _cached_participant_ids(
10275 words_all,
10276 fixations_all,
10277 cache_key=(
10278 frame_fingerprint(words_all),
10279 frame_fingerprint(fixations_all),
10280 ),
10281 )
10282 _refresh_participant_metadata(participants_all)
10284 # UX-37: the dataset is loaded and normalized — one line saying *what*, and
10285 # only when it changes. A rerun re-executes all of this, so an unconditional
10286 # log here would print on every widget touch anywhere in the app.
10287 log_state_change(
10288 "dataset",
10289 (str(data_choice), len(words_all), len(fixations_all)),
10290 "Dataset ready",
10291 source=data_choice,
10292 words=len(words_all),
10293 fixations=len(fixations_all),
10294 )
10296 # VAL-7: does one `trial_id` actually cover several readings? A Trial ID
10297 # mapping that under-specifies concatenates them, and the figure renders as
10298 # an ordinary scanpath with a lot of regressions — nothing looks wrong. Run
10299 # on the *unfiltered* frames: this is a property of the mapping, not of the
10300 # current filter. The full evidence table is in 🔎 Data Inspection; here it
10301 # gets one line, because the column name is the remedy.
10302 # PERF-6: screen a sample by default; the Data page's "Check every trial"
10303 # button sets this flag, which is what asks for the full census.
10304 identity_sample = (
10305 None if st.session_state.get(TRIAL_IDENTITY_FULL_KEY) else TRIAL_IDENTITY_SAMPLE
10306 )
10307 identity_report = _cached_trial_identity_report(
10308 words_all,
10309 fixations_all,
10310 cache_key=(frame_fingerprint(words_all), frame_fingerprint(fixations_all)),
10311 sample_trials=identity_sample,
10312 )
10313 st.session_state["_trial_identity_report"] = identity_report
10314 identity_warning = trial_identity_warning(identity_report)
10315 # BUG-32: an empty (or unjoinable) words frame beside healthy fixations is
10316 # a legitimate *fixations-only* dataset only when no words table was loaded
10317 # at all — otherwise it is a mapping that joins on nothing, and the figure
10318 # just draws without text. A warning, not an error: the fixations are still
10319 # worth drawing, but the silence has to go.
10320 if (st.session_state.get("_active_column_mapping") or {}).get(
10321 "words"
10322 ) and _cached_words_join_nothing(
10323 words_all,
10324 fixations_all,
10325 cache_key=(frame_fingerprint(words_all), frame_fingerprint(fixations_all)),
10326 ):
10327 menu.notices.warning(WORDS_JOIN_NOTHING_WARNING)
10328 # The verdict is raised **once, where the mapping was chosen** — right after
10329 # ✅ Add dataset or ✅ Save changes — rather than as a page-wide banner that
10330 # stood above every view for as long as the dataset was loaded. Both flows
10331 # set `TRIAL_IDENTITY_CHECK_KEY`; the report they are asking about is the one
10332 # just computed above, on the frames those buttons produced.
10333 asked_by = st.session_state.pop(TRIAL_IDENTITY_CHECK_KEY, None)
10334 added = st.session_state.pop(DATASET_ADDED_KEY, None)
10335 if added:
10336 # #374 F30: ✅ Add dataset ended with no word — say what arrived.
10337 st.toast(
10338 dataset_added_message(str(added), words_all, fixations_all),
10339 icon=ICONS["success"],
10340 )
10341 if asked_by and identity_warning:
10342 try:
10343 _trial_identity_alert_dialog(str(asked_by), identity_warning)
10344 except StreamlitAPIException:
10345 # Streamlit allows one dialog per script run, and ❓ Help's three
10346 # (FAQ / About / Tutorials) and the welcome tour are served above
10347 # this point. They cannot normally be armed on the same run as this
10348 # — the flag is set by a button on a screen with no nav reachable —
10349 # but a run that returned early with the flag still armed can. Put
10350 # it back rather than lose the verdict; the next run has no modal
10351 # ahead of it.
10352 st.session_state[TRIAL_IDENTITY_CHECK_KEY] = asked_by
10354 # Trial-level filtering / grouping: narrow by participant, by condition
10355 # (Hunting/Gathering, difficulty, first/repeated reading, correctness), and by
10356 # annotation state (favorites / tags) before anything downstream sees the
10357 # data. The controls now live in the Scanpath tab's Trial Selection panel
10358 # (rendered there via render_trial_filters); here we just read the last
10359 # selection from session_state so filtering stays global across every view.
10360 if dataset_card is not None and dataset_card.steps:
10361 dataset_card.step(len(dataset_card.steps) - 1) # Building the trial list
10362 trial_filters = read_trial_filters()
10363 # BUG-103: each narrowing below makes new frames every rerun while a filter
10364 # is on. `assign_derived` names them by their inputs and settings, so the
10365 # caches downstream are keyed without hashing the whole pool each time.
10366 pool = (words_df, fixations_df)
10367 words_df, fixations_df = filter_trials(
10368 words_df,
10369 fixations_df,
10370 participants=trial_filters["participants"],
10371 metadata=trial_filters["metadata"],
10372 ranges=trial_filters.get("ranges"),
10373 drop_unknown=trial_filters.get("ranges_drop_unknown"),
10374 )
10375 assign_derived(
10376 (words_df, fixations_df),
10377 "filter_trials",
10378 pool,
10379 (
10380 trial_filters["participants"],
10381 trial_filters["metadata"],
10382 trial_filters.get("ranges"),
10383 tuple(trial_filters.get("ranges_drop_unknown") or ()),
10384 ),
10385 )
10386 # DATA-29: a trial-grain metadata narrowing is already `(participant_id,
10387 # trial_id)` keys, so it applies through `filter_to_keys` rather than
10388 # `filter_trials` — the table is never broadcast onto the frames, which is
10389 # DATA-20's rule and the reason this design holds at either grain. `None`
10390 # means no constraint; an empty set legitimately narrows to nothing.
10391 # VIZ-45: the samples table is narrowed by participant like the other two.
10392 # It used to be narrowed only through them (the participant and trial lists
10393 # the filtered words/fixations still held, below), which did nothing on a
10394 # raw-gaze-only dataset. The condition filters reach it only when it is the
10395 # dataset's only table — they are offered from its own columns then
10396 # (`tabs.render_single_trial_tab` → `_render_filters`); beside fixations
10397 # they name *their* columns, and a raw-gaze column of the same name (a
10398 # `text_id` that merely mirrors the trial id) means something else.
10399 samples_only_dataset = words_all.empty and fixations_all.empty
10400 trialmeta_keys = trial_filters.get("trial_keys")
10401 raw_gaze_df = _narrowed_raw_gaze(
10402 raw_gaze_df,
10403 participants=trial_filters["participants"],
10404 metadata=trial_filters["metadata"] if samples_only_dataset else None,
10405 ranges=trial_filters.get("ranges") if samples_only_dataset else None,
10406 trial_keys=trialmeta_keys,
10407 drop_unknown=trial_filters.get("ranges_drop_unknown")
10408 if samples_only_dataset
10409 else None,
10410 )
10411 if trialmeta_keys is not None:
10412 pool = (words_df, fixations_df)
10413 words_df, fixations_df = filter_to_keys(words_df, fixations_df, trialmeta_keys)
10414 assign_derived((words_df, fixations_df), "filter_to_keys", pool, trialmeta_keys)
10415 # BUG-12: the raw-gaze samples table has to travel through the same
10416 # annotation filter as words + fixations, or a sample row for an unstarred
10417 # trial survives "⭐ Favorites only" — which also kept the all-three-empty
10418 # guard below from ever firing, leaving the UX-7 guidance panel unreachable.
10419 raw_gaze_scoped = raw_gaze_df
10420 if (
10421 trial_filters["favorites_only"]
10422 or trial_filters["required_tags"]
10423 or trial_filters["excluded_tags"]
10424 ) and not (fixations_df.empty and words_df.empty and raw_gaze_scoped.empty):
10425 # Trials live in fixations normally; for words-only datasets the words
10426 # frame carries them, and for raw-gaze-only ones the samples table —
10427 # union all three so every frame's trials get judged by the filter.
10428 present_keys = (
10429 trial_keys(words_df)
10430 | trial_keys(fixations_df)
10431 | trial_keys(raw_gaze_scoped)
10432 )
10433 kept = set(
10434 filter_keys(
10435 list(present_keys),
10436 favorites_only=trial_filters["favorites_only"],
10437 required_tags=trial_filters["required_tags"],
10438 excluded_tags=trial_filters["excluded_tags"],
10439 )
10440 )
10441 pool = (words_df, fixations_df, raw_gaze_scoped)
10442 words_df, fixations_df = filter_to_keys(words_df, fixations_df, kept)
10443 raw_gaze_scoped = filter_frame_to_keys(raw_gaze_scoped, kept)
10444 assign_derived(
10445 (words_df, fixations_df, raw_gaze_scoped), "filter_to_keys", pool, kept
10446 )
10448 # Apply filters (participant/trial/text selection). For a raw-gaze-only
10449 # dataset (no words/fixations) derive the participant/trial options from the
10450 # raw gaze so it isn't filtered away (filter_raw_gaze drops on empty lists).
10451 filters = default_filters(
10452 words_df, fixations_df if not fixations_df.empty else raw_gaze_scoped
10453 )
10454 words_filtered, fixations_filtered = filter_data(words_df, fixations_df, filters)
10455 assign_derived(
10456 (words_filtered, fixations_filtered),
10457 "filter_data",
10458 (words_df, fixations_df),
10459 filters,
10460 )
10462 # The samples of the trials in the pool — and of every trial only the raw
10463 # gaze has, which no filter on the other two tables can speak for (VIZ-45;
10464 # this used to keep only the participants and trials those tables listed,
10465 # so a samples-only trial was dropped from any dataset with fixations).
10466 if not raw_gaze_scoped.empty:
10467 raw_gaze_filtered = raw_gaze_in_pool(
10468 raw_gaze_scoped,
10469 words_all,
10470 fixations_all,
10471 words_filtered,
10472 fixations_filtered,
10473 )
10474 if raw_gaze_filtered.empty:
10475 # Informational, not an error: the loaded raw-gaze samples just
10476 # don't cover any trial in the current filter (raw gaze typically
10477 # exists for only a subset of trials). The overlay is optional.
10478 menu.notices.caption(
10479 f"{ICONS['info']} The loaded raw-gaze samples ({len(raw_gaze_all):,} rows) don't "
10480 "overlap the current trial filter, so the raw-gaze overlay is "
10481 "unavailable here."
10482 )
10483 else:
10484 raw_gaze_filtered = pd.DataFrame()
10486 # Check for empty data after filtering. A single empty frame is fine
10487 # (words-only / fixations-only / raw-gaze-only datasets); all empty means the
10488 # filters removed everything.
10489 if words_filtered.empty and fixations_filtered.empty and raw_gaze_filtered.empty:
10490 _render_empty_after_filtering(
10491 words_all,
10492 fixations_all,
10493 trial_filters,
10494 pool_filter_frames(words_all, fixations_all, raw_gaze_all),
10495 )
10496 _finish_page()
10497 _render_datasets_table(words_all, fixations_all, raw_gaze_all)
10498 _end_loading()
10499 return
10501 # Build trial combinations for selection UI — from fixations normally, then
10502 # words (words-only datasets), then raw gaze (raw-gaze-only datasets), plus
10503 # any trial only the raw gaze has (VIZ-45, `utils.combo_source`).
10504 combos, _, _ = build_combo_options(
10505 combo_source(fixations_filtered, words_filtered, raw_gaze_filtered)
10506 )
10507 # DATA-20 §3 — the *one* place the participant table is joined onto anything.
10508 # `combos` is one row per trial (tens to thousands), so this is the cheap
10509 # projection the item asks for rather than a broadcast across every fixation
10510 # — and it is enough: trial sorting, the trial labels and everything else
10511 # downstream discovers its columns from this frame, with no allowlist to
10512 # extend per surface.
10513 combos = metadata_mod.project(metadata_mod.active(), combos)
10514 # DATA-29 §3 — and the one place the *trial* table is joined, onto the same
10515 # small frame. Everything downstream (the chip picker, trial sorting, Data
10516 # Inspection, export) discovers its columns from `combos`, so this single
10517 # left-join is what makes a trial field behave like a field in the data —
10518 # again without broadcasting the table onto words or fixations.
10519 combos = metadata_mod.project_trials(metadata_mod.active_trials(), combos)
10520 # And the text table, the third grain, onto the same small frame — same
10521 # reasoning again.
10522 combos = metadata_mod.project_texts(metadata_mod.active_texts(), combos)
10524 # Land a shared/deep link on its exact `?trial_id=` (once) now that combos
10525 # exist — see _apply_url_trial_selection. Runs before the rail/tab widgets
10526 # render so the seeded selection is picked up as their initial value.
10527 # A link the pool cannot answer — its reader filtered out, or a trial id
10528 # several readers share with none named — is reported, and not retried.
10529 if missed_link := _apply_url_trial_selection(combos):
10530 menu.notices.warning(missed_link, icon=ICONS["warning"])
10531 # Same hop, from inside the app: a "go to this trial" button in a Corpus
10532 # Analysis table parks its request in a callback (before combos exist) and
10533 # it is applied here — see url_state.request_trial (ENG-36).
10534 # A reading the pool cannot answer — its reader filtered out, say — is
10535 # reported, never replaced by another reader's trial of the same name.
10536 if missed := _apply_pending_trial_selection(combos):
10537 menu.notices.warning(missed, icon=ICONS["warning"])
10539 # Restore settings from an uploaded settings file BEFORE the rail widgets
10540 # render, so they pick up the saved values (see _apply_url_preset for the
10541 # same preset-then-render mechanism). The uploader is 🔗 Share → File's; its
10542 # file persists across reruns.
10543 _apply_uploaded_plot_config(combos, fixations_filtered)
10545 # Canvas and visualization controls (the Scanpath rail). For a dataset whose
10546 # only gaze is samples, size the canvas from the gaze extent and turn the
10547 # raw-gaze layer on — it's the only gaze layer there, so the plot would
10548 # otherwise show no gaze at all (VIZ-45). Decided once per dataset, on the
10549 # unfiltered frames; a link that names the layer decides it instead.
10550 raw_gaze_source_key = (data_choice, st.session_state.get("public_dataset_choice"))
10551 seed_raw_gaze_default(
10552 st.session_state,
10553 raw_gaze_source_key,
10554 samples_only=fixations_all.empty and not raw_gaze_all.empty,
10555 link_names_layer=lambda: link_sets(_RAW_GAZE_LAYER_KEY),
10556 )
10557 # VIZ-31: the monitor/font/background panel moved out of the sidebar into the
10558 # Scanpath rail, so it is *resolved* here (no widgets) and *rendered* later,
10559 # inside the rail, via the `canvas_renderer` below. Resolving first is what
10560 # keeps the Corpus view — which has no rail — on the same canvas + typography.
10561 canvas_geometry_frame = (
10562 fixations_filtered if not fixations_filtered.empty else raw_gaze_filtered
10563 )
10564 (
10565 canvas_width,
10566 canvas_height,
10567 base_font_size,
10568 font_family,
10569 line_spacing,
10570 scale_text_to_boxes,
10571 ) = seed_canvas_state(words_filtered, canvas_geometry_frame, data_choice)
10573 def canvas_renderer(
10574 slot,
10575 text_host=None,
10576 *,
10577 render_text: bool = True,
10578 text_disabled: bool = False,
10579 ) -> None:
10580 """Render the canvas/text controls into the rail, in two places.
10582 UX-81 split the panel between two sections: the screen half into
10583 ``slot`` (📐 Figure & canvas) and the typography half into ``text_host``
10584 (📄 Stimulus → Text). One call, so each widget is created exactly once.
10585 With no ``text_host`` (the Corpus style panel) the typography rows are
10586 titled *Text* themselves, since no *Text* row precedes them there.
10587 """
10588 render_canvas_controls(
10589 words_filtered,
10590 canvas_geometry_frame,
10591 data_choice,
10592 slot=slot,
10593 bare=True,
10594 text_host=text_host,
10595 render_text=render_text,
10596 text_disabled=text_disabled,
10597 text_section=None if text_host is not None else "Text",
10598 )
10600 # The visualization controls moved out of the sidebar into the Scanpath
10601 # screen's right-hand rail (tabs.render_single_trial_tab renders them via
10602 # controls.render_plot_controls with host=rail). The other views — and the Save &
10603 # restore panel below — still need the resolved settings, so read them from
10604 # session_state without rendering any widgets; the rail's widgets are the
10605 # source of truth and write the same keys.
10606 viz_settings = viz_settings_from_state(
10607 fixations_filtered, base_font_size, words=words_filtered
10608 )
10610 # Whole-dataset combos for the Bulk Export tab's "Export the whole dataset"
10611 # option, mirroring how `combos` is built from the filtered frames.
10612 combos_all, _, _ = build_combo_options(
10613 combo_source(fixations_all, words_all, raw_gaze_all)
10614 )
10616 # UX-166: the load is done; the skeleton stays until the view has drawn its
10617 # controls and its first slow region opens (see `loading.card`).
10618 _finish_dataset_card(dataset_card, data_choice)
10620 # Render tabbed interface. Animation is now a checkbox inside the Scanpath
10621 # Visualization tab (no separate Animated Scanpath tab); Bulk Export has its
10622 # own tab. Raw Data + Data Statistics are merged into Data Inspection.
10623 # Dispatch the active view (top nav). Only one view body renders per run
10624 # — the keyed nav widget persists the selection across reruns, so no JS hack
10625 # is needed (unlike st.tabs). render_single_trial_tab writes _share_selection,
10626 # which 🔗 Share → File reads for the trial it records.
10627 # UX-37: the three things a log reader wants to correlate a slow rerun with
10628 # — which view, which trial, how narrow the pool is. One line each, only on
10629 # change (see `log_state_change`).
10630 log_state_change("view", active_view, "View", view=active_view)
10631 log_state_change(
10632 "filters",
10633 (len(words_filtered), len(fixations_filtered), len(combos)),
10634 "Filters applied",
10635 trials=len(combos),
10636 words=len(words_filtered),
10637 fixations=len(fixations_filtered),
10638 )
10640 if data_view:
10641 loading.release_page() # UX-166: the Data page's card sits above its table
10642 # DATA-26 — fill the page reserved before the load. Everything above the
10643 # dispatch already landed in its slot (source picker · description ·
10644 # options · data location · wizard · mode-A mapping panels); what is left
10645 # needs the loaded frames, so it renders here.
10646 #
10647 # UX-54's dataset table is the first of those: its counts for the open
10648 # dataset are this run's frames, which do not exist until the load has
10649 # happened. Unfiltered on purpose — the table describes the *dataset*,
10650 # not what the current Narrow-by left standing. UX-203: while a trial
10651 # filter is on, or the demo stands in for the open corpus, the counts
10652 # below it (📊 Stats) differ from the row's, so both say which they are.
10653 trials_filtered = has_active_trial_filters()
10654 _render_datasets_table(
10655 words_all,
10656 fixations_all,
10657 raw_gaze_all,
10658 scope_note=dataset_table_scope_note(
10659 filtered=trials_filtered,
10660 stand_in_for=(
10661 _dataset_display_name(
10662 str(st.session_state.get("data_source_choice") or data_choice)
10663 )
10664 if st.session_state.get(_PLACEHOLDER_SHOWN_KEY)
10665 else None
10666 ),
10667 ),
10668 )
10669 # UX-135 — one numbered headline over the whole first part, drawn into
10670 # the slot reserved above the description. Everything from here to the
10671 # metadata tables is that part; the mapping no longer titles itself,
10672 # any more than the add screen's mapping rows do.
10673 _editor_part(editor_part_data_slot, "edit_data")
10674 # UX-135 — Recording setup renders into its own numbered part instead of
10675 # trailing the mapping form. Same renderer, same state, different host.
10676 recording_body = _editor_part(setup_recording_slot, "edit_setup")
10677 with mapping_body_slot:
10678 _render_column_mapping_section(
10679 editor_rendered=mapping_editor_rendered,
10680 uploads_host=editor_uploads_slot,
10681 setup_host=recording_body,
10682 words=words_all,
10683 fixations=fixations_all,
10684 )
10685 with _editor_part(setup_identity_slot, "edit_identity"):
10686 render_trial_identity_section()
10687 # ``setup_stimulus_slot`` is filled earlier because its values resolve
10688 # image paths before filtering and plotting; its reserved position is
10689 # what puts it after Trial identity on screen regardless (creation order
10690 # is screen order, so where a slot is *filled* need not agree).
10691 with setup_metadata_slot:
10692 # UX-130 r2: the *add screen's* three metadata rows, not three
10693 # side-by-side panels. UX-114 put them in one row of three columns
10694 # (an improvement on the stacked full-width blocks before it) and
10695 # UX-127 then gave the add screen a different shape again — one
10696 # "left = upload, right = mapping" row per table, under a small
10697 # **Metadata** heading, matching the Fixations/AOI/Raw gaze rows
10698 # above them. The two screens ask the same question of the same
10699 # dataset, so they draw it the same way; this is the same
10700 # `render_*_metadata_section` renderers in the same `_META_ROW_W`
10701 # split, under the same `wiz_map_*` key prefix so `styles.py`'s
10702 # own `[class*="st-key-wiz_map_…"]` rules apply verbatim — with an
10703 # `_edit` suffix, since a container key may be used once per run
10704 # and the offscreen editor is built even while the wizard is open.
10705 #
10706 # The *unfiltered* pool feeds every report (participants_all /
10707 # combos_all): the join describes the dataset, not whatever the
10708 # current trial filters left standing. `live_join` stays on, unlike
10709 # the wizard's (UX-116) — here the dataset exists, so the join is a
10710 # real answer rather than a provisional one.
10711 from scanpath_studio.controls import inline_field_label
10712 from scanpath_studio.wizard import _META_ROW_W
10714 meta_heading_row = st.columns(
10715 [_META_ROW_W[0], 1 - _META_ROW_W[0]], gap="small"
10716 )
10717 inline_field_label(
10718 meta_heading_row[0].container(key="wiz_map_meta_heading_edit"),
10719 "Metadata",
10720 "Optional per-participant, per-trial and per-text tables. Once "
10721 "attached, their columns behave like fields in the data: "
10722 "filters, trial labels, sorting, inspection and export.",
10723 emphasis=True,
10724 )
10725 for slug, renderer, ids in (
10726 ("participant", render_participant_metadata_section, participants_all),
10727 # DATA-29: same reasoning one grain down, and DATA-TBD one more.
10728 ("trial", render_trial_metadata_section, combos_all),
10729 (
10730 "text",
10731 render_text_metadata_section,
10732 metadata_mod.text_keys(combos_all),
10733 ),
10734 ):
10735 block = st.container(key=f"wiz_map_block_meta_{slug}_edit")
10736 row = block.columns(_META_ROW_W, gap="small")
10737 renderer(
10738 ids,
10739 host=row[1],
10740 upload_host=row[0].container(
10741 key=f"wiz_map_upload_meta_{slug}_edit"
10742 ),
10743 )
10744 st.markdown(
10745 '<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True
10746 )
10747 # UX-106 — and the screen's one commit at its foot, in the slot
10748 # reserved after every section it saves.
10749 render_dataset_editor_footer(editor_footer_slot)
10750 if mapping_editor_rendered:
10751 _render_builtin_editor_footer(editor_footer_slot)
10752 elif str(st.session_state.get("data_source_choice") or "") not in (
10753 st.session_state.get("_datasets") or {}
10754 ) and data_choice not in (UPLOAD_CHOICE, AUTHOR_CHOICE, MANUAL_SAMPLE_CHOICE):
10755 # A source with no mapping panels (the synthetic trial, a server
10756 # bundle) still has a name, a description and metadata tables to
10757 # save — and they wait for Save like everything else.
10758 _render_builtin_editor_footer(editor_footer_slot, mapping=False)
10759 with setup_body_slot:
10760 st.divider()
10761 active_token = str(
10762 st.session_state.get("data_source_choice") or data_choice
10763 )
10764 # UX-137 — the open dataset's own prose is one sentence under the
10765 # heading, with the rest behind a ❔. It used to be a full ℹ️ About
10766 # section *above* this heading: a second subheader, the description,
10767 # the corpus home link, the coordinate-provenance sentence and a
10768 # six-row published-vs-loaded table, all standing between the user
10769 # and the counts they came for. UX-174 r2 put Rename on the heading
10770 # and Edit on the description line, off the table's rows; Edit now
10771 # sits under the overview, above the subtabs.
10772 render_dataset_inspection_head(active_token)
10773 # DATA-67 — what the dataset supports, before any trial filter:
10774 # the first thing a newly added dataset's overview answers.
10775 render_dataset_capabilities(
10776 words_all, fixations_all, raw_gaze_all, filtered=trials_filtered
10777 )
10778 # Values that parsed but cannot be right (negative durations,
10779 # infinite positions, empty word boxes) — counted, not removed.
10780 render_data_health(
10781 words_all, fixations_all, raw_gaze_all, filtered=trials_filtered
10782 )
10783 render_dataset_edit_button(active_token)
10784 # Keyed wrapper → the stable `.st-key-…` selector the "Load and
10785 # verify a dataset" tutorial spotlights (it kept its name across the
10786 # move off the Scanpath subtab bar).
10787 with st.container(key="tutorial_data_inspection"):
10788 render_data_inspection_tab(
10789 words_filtered,
10790 fixations_filtered,
10791 raw_gaze_filtered,
10792 annotation_trials=_annotation_trials(combos_all),
10793 # #374 F5: each trial as the trial picker writes it.
10794 annotation_trial_labels=_annotation_trial_labels(combos_all),
10795 # What the Scanpath picker can open — an annotation row's
10796 # Open explains a trial the filters hide.
10797 open_trials=_annotation_trials(combos),
10798 # DATA-48: the dataset whose annotations these are — the
10799 # demo's while it stands in for a missing corpus, as in
10800 # the Export bundle.
10801 dataset_name=_dataset_display_name(
10802 _annotations_dataset(active_token)
10803 ),
10804 scope=data_scope_text(combos, combos_all, words_all, fixations_all),
10805 )
10806 elif active_view == _VIEW_CORPUS:
10807 with view_area:
10808 # UX-25: Corpus Analysis has no "Filter by" row, so the picker gets
10809 # its own compact row at the top of the page — it stays reachable on
10810 # every view.
10811 _ds_col, pool_col = st.columns([2, 5], vertical_alignment="bottom")
10812 render_data_source_picker(host=_ds_col)
10813 # UX-198: the pool the analysis reads — and the filters behind it —
10814 # beside the dataset, where Scanpath keeps its own filter funnel.
10815 pool = render_analysis_pool_bar(
10816 pool_col,
10817 words_all=words_all,
10818 fixations_all=fixations_all,
10819 raw_gaze_all=raw_gaze_all,
10820 combos=combos,
10821 combos_all=combos_all,
10822 )
10823 # AN-34: what each table's recipe shares — the dataset by the name
10824 # the picker shows (and its stable token), and the pool above.
10825 source_token = str(
10826 st.session_state.get("data_source_choice") or data_choice
10827 )
10828 recipe_context = {
10829 **pool,
10830 "dataset": {
10831 "name": _dataset_display_name(source_token),
10832 "source": source_token,
10833 },
10834 }
10835 with st.container(key="tutorial_corpus_analysis"):
10836 render_corpus_analysis_tab(
10837 words_filtered,
10838 fixations_filtered,
10839 canvas_width=canvas_width,
10840 canvas_height=canvas_height,
10841 base_font_size=base_font_size,
10842 font_family=font_family,
10843 viz_settings=viz_settings,
10844 line_spacing=line_spacing,
10845 scale_text_to_boxes=scale_text_to_boxes,
10846 canvas_renderer=canvas_renderer,
10847 has_raw_gaze=not raw_gaze_filtered.empty,
10848 recipe_context=recipe_context,
10849 )
10850 else:
10851 # The Scanpath view renders the viz controls itself (right rail) and
10852 # writes the global_* keys. Share is a subtab of this view — passed in as
10853 # a renderer so the page owns its subtab bar — and its File section
10854 # re-reads those keys when it draws, which is after the rail has.
10856 def _render_share_settings_file() -> None:
10857 """🔗 Share → File (UX-179): the settings file of the figure on screen.
10859 Resolved only when the File section is drawn — the trial from
10860 ``_share_selection`` (written by the Scanpath view before its
10861 subtabs), the figure settings from the rail's live keys.
10862 """
10863 live = viz_settings_from_state(
10864 fixations_filtered, base_font_size, words=words_filtered
10865 )
10866 selection = st.session_state.get("_share_selection") or {}
10867 pid = str(selection.get("participant_id") or "")
10868 trial = str(selection.get("trial_id") or "")
10869 screen = selection.get("screen_id")
10870 # The trial first (position-indexed), then its screen: `extract_part`
10871 # compares strings row by row, which over a whole sample-level
10872 # raw-gaze frame is seconds per call.
10873 trial_raw_gaze = (
10874 extract_trial(raw_gaze_filtered, pid, trial)
10875 if pid and trial and not raw_gaze_filtered.empty
10876 else pd.DataFrame()
10877 )
10878 if screen is not None and SCREEN_ID in trial_raw_gaze.columns:
10879 trial_raw_gaze = extract_part(trial_raw_gaze, pid, trial, screen)
10880 figure_settings = _build_figure_settings(live, not trial_raw_gaze.empty)
10881 figure_settings["raw_gaze"] = (
10882 trial_raw_gaze if not trial_raw_gaze.empty else None
10883 )
10884 figure_settings["line_spacing"] = line_spacing
10885 figure_settings["scale_text_to_boxes"] = scale_text_to_boxes
10886 render_settings_file(
10887 pid,
10888 trial,
10889 canvas_width,
10890 canvas_height,
10891 live["x_field"],
10892 live["y_field"],
10893 figure_settings,
10894 live,
10895 base_font_size,
10896 trial_raw_gaze,
10897 font_family=font_family,
10898 )
10900 with view_area:
10901 render_single_trial_tab(
10902 words_filtered,
10903 fixations_filtered,
10904 combos,
10905 canvas_width=canvas_width,
10906 canvas_height=canvas_height,
10907 base_font_size=base_font_size,
10908 font_family=font_family,
10909 raw_gaze=raw_gaze_filtered,
10910 line_spacing=line_spacing,
10911 scale_text_to_boxes=scale_text_to_boxes,
10912 combos_all=combos_all,
10913 words_all=words_all,
10914 fixations_all=fixations_all,
10915 raw_gaze_all=raw_gaze_all,
10916 share_renderer=lambda visible: _render_share_body(
10917 data_choice,
10918 settings_file=_render_share_settings_file,
10919 visible=visible,
10920 ),
10921 data_source_renderer=render_data_source_picker,
10922 canvas_renderer=canvas_renderer,
10923 )
10925 # UX-166: a view that never reached a slow region (no trial selected, an
10926 # empty view) still takes the skeleton down.
10927 loading.release_page()
10929 # UX-65 — ❓ Help is a *menu* in the nav now, not a page of buttons: each
10930 # entry arms the same dialog its button used to (menu._arm_help_action), and
10931 # the dialogs themselves are untouched. Nothing to fill here anymore — only
10932 # the tutorial chooser's context, which is data, not a widget, and has to be
10933 # stashed on every run so the dialog can open over any view.
10934 #
10935 # 📚 Documentation left with the buttons: `st.Page` cannot be a URL, and the
10936 # UX-62 wordmark beside the nav already opens the docs site.
10937 tutorial_context = build_tutorial_context(
10938 words_filtered, fixations_filtered, combos
10939 )
10940 if len(combos) != len(combos_all):
10941 # #374 F31: while a filter narrows the pool, a cheap snapshot of the
10942 # whole dataset lets the chooser say the filter is why a tutorial
10943 # can't start (from the trial list alone — no frame is regrouped).
10944 unfiltered = build_tutorial_context(None, None, combos_all)
10945 unfiltered["has_words"] = words_all is not None and not words_all.empty
10946 unfiltered["has_fixations"] = (
10947 fixations_all is not None and not fixations_all.empty
10948 )
10949 tutorial_context["unfiltered"] = unfiltered
10950 stash_tutorial_context(tutorial_context)
10951 # Persist after all view/menu widgets have written their current values.
10952 # The helper fingerprints the session and is a no-op on unchanged reruns.
10953 save_local_state(st.session_state, app_url)
10954 # …then draw *Saved on this computer*, so it reports the write that just
10955 # happened rather than the previous run's.
10956 _finish_page()
10959if __name__ == "__main__":
10960 from scanpath_studio.crash_report import run_app
10962 run_app()