Coverage for scanpath_studio/compare_source.py: 89%
169 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""The *second* dataset a comparison can draw scanpath B from (CMP-8 §2).
3Compare mode used to pick B out of the same loaded corpus as A. This module is
4what lets it reach a different one — a PoTeC reader beside a OneStop reader, or
5the same text read under two corpora.
7**The one hard constraint: nothing here may render.** The app's public-corpus
8loaders (`app._load_public_dataset`) draw directory inputs, *Expected files*
9layouts and ⬇ Download buttons; none of that can appear inside the compare
10picker, which is a selectbox in the middle of the plot column. So this module
11reads the *location state those loaders already wrote* (`<prefix>_dir`) and goes
12straight to the widget-free `datasets.load_*` functions. A corpus whose location
13has never been set is offered **disabled** with a reason, never loaded blind.
15`app` is imported lazily inside the functions, not at module scope: `app`
16imports `tabs`, `tabs` imports this, so a module-level import would close the
17cycle (the same reason `wizard.py` is imported lazily by `app`).
18"""
20from __future__ import annotations
22from collections.abc import Mapping
23from dataclasses import dataclass, field
25import pandas as pd
26import streamlit as st
28from . import progress
29from .column_names import ColumnNames
30from .constants import (
31 DEMO_CHOICE,
32 EYEGENBENCH_DEFAULT_DIR,
33 MULTIPLEYE_DEFAULT_DIR,
34 ONESTOP_PUBLIC_DEFAULT_DIR,
35 POTEC_DEFAULT_DIR,
36 SYNTHETIC_CHOICE,
37 onestop_regime_for_choice,
38)
39from .data import adopt_source, stamp_source, vouch_for_frames
40from .experimental_setup import Provenance, SetupSnapshot
41from .session_keys import COMPARE_SOURCE_STATE_KEY
43#: The picker's "stay in this dataset" entry — compare mode's behaviour before
44#: CMP-8, and still the default.
45THIS_DATASET = "This dataset"
47#: Session key holding the picked secondary source name. Re-exported from
48#: `session_keys` rather than spelled again: it is a deep-link-seeded key, and
49#: two literals for one wire-format name is exactly the drift `session_keys.py`
50#: exists to prevent.
51COMPARE_SOURCE_KEY = COMPARE_SOURCE_STATE_KEY
53#: `<key_prefix>_dir` session keys written by `app._dataset_dir_input`, plus the
54#: default each loader passes it. Read-only here: this module never writes a
55#: location, it only reports whether one is usable.
56_MULTIPLEYE_LABEL_HINT = "MultiplEYE"
57_POTEC_LABEL_HINT = "PoTeC"
60@dataclass(frozen=True, eq=False)
61class SecondaryDataset:
62 """One loaded comparison source: normalized frames plus what they were shown on.
64 ``eq=False`` because the frames make dataclass equality ambiguous (pandas
65 raises on a truth-valued comparison), and nothing compares these.
66 """
68 name: str
69 words: pd.DataFrame
70 fixations: pd.DataFrame
71 combos: pd.DataFrame
72 setup: SetupSnapshot
73 composite_trial_columns: tuple[str, ...] = field(default=())
74 #: VIZ-48: the source's normalized raw gaze, when it carries any — an
75 #: upload's, or the bundled demo's. The public corpora ship none.
76 raw_gaze: pd.DataFrame | None = None
77 #: DATA-66: B's column-name map per table — a stored upload's own; empty for
78 #: a public corpus or the demo until their loaders return one (phase 4).
79 column_names: Mapping[str, ColumnNames] = field(default_factory=dict)
82def _resolved_dir(key: str, default_dir: str) -> str:
83 """The directory a public-corpus loader would read, without rendering it.
85 Mirrors `app._dataset_dir_input`'s return value: the session key the text
86 input wrote, else the loader's default, resolved against the project root.
87 """
88 from scanpath_studio import app
90 raw = str(st.session_state.get(key) or "").strip() or default_dir
91 if app.data_root() and not app.local_filesystem_enabled():
92 # S2: on a shared deployment the path box isn't rendered at all and the
93 # location comes from the server's environment — the same rule the
94 # loader itself follows.
95 return str(app.data_root())
96 return app._resolve_data_dir(raw)
99def _public_location(label: str) -> tuple[str, dict]:
100 """``(root, loader kwargs)`` for a `PUBLIC_DATASET_REGISTRY` label.
102 The kwargs are the *user's current* source options (OneStop's variant /
103 regime / parts, MultiplEYE's fixation source) — the same session keys the
104 the Compare-with widgets own, read rather than re-rendered.
105 """
106 from scanpath_studio import app, datasets
108 # DATA-27 (Task 11R): a prepared benchmark corpus is one registry entry that
109 # names the corpus inside the bundle, so it dispatches on `benchmark_dataset`
110 # — **before** the label-substring branches below, which a harmonised
111 # "PoTeC …" / "OneStop …" label would otherwise match and send to the native
112 # loader.
113 spec = app.public_dataset_registry().get(label) or {}
114 if dataset := spec.get("benchmark_dataset"):
115 return _resolved_dir("eyegenbench_dir", EYEGENBENCH_DEFAULT_DIR), {
116 "dataset": dataset,
117 }
118 if regime := onestop_regime_for_choice(label):
119 # DATA-63: one dataset per regime, every part, from the public release.
120 # UX-184: the box's own default — under the Download folder.
121 default = app._download_target(ONESTOP_PUBLIC_DEFAULT_DIR)
122 return _resolved_dir("onestop_public_dir", default), {
123 "variant": "public",
124 "regime": regime,
125 "parts": tuple(datasets.onestop_regime_parts(regime)),
126 }
127 if _POTEC_LABEL_HINT in label:
128 return _resolved_dir("potec_dir", app._download_target(POTEC_DEFAULT_DIR)), {}
129 if _MULTIPLEYE_LABEL_HINT in label:
130 return _resolved_dir("multipleye_dir", MULTIPLEYE_DEFAULT_DIR), {
131 "fixation_source": str(
132 st.session_state.get("multipleye_fixation_source") or "scanpaths"
133 ),
134 }
135 return "", {}
138def _public_ready(label: str) -> tuple[bool, str]:
139 """Whether a public corpus can be loaded *silently*, and why not if it can't.
141 Uses the existing readiness helpers — `datasets.potec_present` /
142 `onestop_present` / `multipleye_inventory` — so a corpus is never offered as
143 B unless the very same check the main source picker runs says its files are
144 there. A download is deliberately never triggered from here.
146 Cached on the resolved location + source options: this runs for *every*
147 registry entry on every rerun that Compare is on, and `multipleye_inventory`
148 walks each session directory (`app._cached_multipleye_inventory` wraps the
149 same call for the same reason). Location changes bust the key.
150 """
151 root, kwargs = _public_location(label)
152 return _public_ready_cached(
153 label, root, tuple(sorted(kwargs.items())), _short_name(label)
154 )
157def _short_name(label: str) -> str:
158 """What to call this corpus in a disabled entry's hint.
160 The registry's own ``short`` name, not the label's prefix: two entries can
161 share one prefix — the native and the harmonised PoTeC both split to
162 ``"PoTeC"`` — so a prefix-derived hint told the user to go and open one of
163 two entries it couldn't tell apart (M12).
164 """
165 from scanpath_studio import app
167 spec = app.public_dataset_registry().get(label) or {}
168 return str(spec.get("short") or "").strip() or label.split(" — ")[0]
171@st.cache_data(show_spinner=False)
172def _public_ready_cached(
173 label: str, root: str, options: tuple, short: str = ""
174) -> tuple[bool, str]:
175 from scanpath_studio import datasets
177 kwargs = dict(options)
178 short = short or label.split(" — ")[0]
179 if not root:
180 return False, f"{short} has no data folder yet — open it once first."
181 hint = f"Open {short} as the main dataset once to download or locate it."
182 try:
183 if dataset := kwargs.get("dataset"):
184 from scanpath_studio.eyegenbench import eyegenbench_present
186 present = eyegenbench_present(root, dataset)
187 elif onestop_regime_for_choice(label):
188 present = datasets.onestop_present(
189 root,
190 regime=kwargs["regime"],
191 parts=list(kwargs["parts"]),
192 variant=kwargs["variant"],
193 )
194 elif _POTEC_LABEL_HINT in label:
195 present = datasets.potec_present(root)
196 elif _MULTIPLEYE_LABEL_HINT in label:
197 sessions, _ = datasets.multipleye_inventory(
198 root, fixation_source=kwargs["fixation_source"]
199 )
200 present = bool(sessions)
201 else:
202 return False, f"{short} isn't loadable as a comparison dataset."
203 except (OSError, ValueError, KeyError):
204 # `KeyError` because a manifest row is data from a file on disk and can
205 # be malformed — a row with no `name` used to escape this catch and take
206 # the whole app down through the compare-B enumeration, which runs over
207 # *every* registry entry on every rerun Compare is on (I2). The nameless
208 # row is now skipped at the source too (`eyegenbench.entry_name`); this
209 # is the belt to that braces, since the same catch covers four loaders'
210 # readiness probes and only one of them has been hardened.
211 present = False
212 return (True, "") if present else (False, hint)
215def secondary_dataset_options(
216 *, exclude: str | None = None
217) -> list[tuple[str, bool, str]]:
218 """Every source compare mode could draw B from: ``(name, ready, why_not)``.
220 Stored uploads, the bundled demo and the synthetic trial are always ready —
221 they are in memory or in the package. Public corpora are *always offered* but
222 ready only when their files are already where the main picker last looked;
223 an unready entry renders disabled with ``why_not`` rather than disappearing,
224 so the capability is discoverable instead of mysteriously absent.
226 The ``$ONESTOP_DATA_DIR`` **server bundle is deliberately not offered.**
227 `data.load_onestop_server_bundle` is sub-second only when it is given a
228 participant to load a per-pid shard for; without one it falls back to the
229 full CSV exports — its own docstring says ~3 min and ~60 GB for the L2
230 cohort. The picker has no participant to give at the moment it builds its
231 options, so offering the bundle would mean blocking the whole app for
232 minutes, and possibly OOM-ing the server, to draw one comparison trial. The
233 same corpus is reachable as the public *OneStop* entry below.
235 ``exclude`` drops one name (the active source — comparing a dataset with
236 itself is what `THIS_DATASET` already means).
237 """
238 from scanpath_studio import app
240 options: list[tuple[str, bool, str]] = [
241 (name, True, "") for name in sorted(st.session_state.get("_datasets") or {})
242 ]
243 options.append((DEMO_CHOICE, True, ""))
244 options.append((SYNTHETIC_CHOICE, True, ""))
245 if app.public_datasets_enabled():
246 options.extend(
247 (label, *_public_ready(label)) for label in app.public_dataset_registry()
248 )
249 return [option for option in options if option[0] != exclude]
252@st.cache_data(show_spinner=False) # UX-168: B's dataset card covers this.
253def _load_public_frames(
254 label: str, root: str, options: tuple
255) -> tuple[pd.DataFrame, pd.DataFrame, dict]:
256 """Normalized frames for a public corpus, keyed on its location + options,
257 and its column-name map per table, as payloads (DATA-66).
259 Goes through the `datasets.load_*` entry points (which normalize internally
260 via `api.load_scanpath_data`), never `app.prepare_data` — that one takes a
261 ``mapping_host`` and renders the column-mapping panels.
262 """
263 from scanpath_studio import datasets
265 kwargs = dict(options)
266 # The app works in the internal names; DATA-66: the corpus' own names come
267 # back beside the frames (`ScanpathData.column_names`), for B's labels.
268 if dataset := kwargs.get("dataset"):
269 from scanpath_studio.eyegenbench import load_eyegenbench
271 data = load_eyegenbench(root, dataset=dataset, names="canonical")
272 elif onestop_regime_for_choice(label):
273 data = datasets.load_onestop(
274 root,
275 regime=kwargs["regime"],
276 parts=list(kwargs["parts"]),
277 variant=kwargs["variant"],
278 names="canonical",
279 )
280 elif _POTEC_LABEL_HINT in label:
281 data = datasets.load_potec(root, names="canonical")
282 else:
283 data = datasets.load_multipleye(
284 root, fixation_source=kwargs["fixation_source"], names="canonical"
285 )
286 # BUG-103: B's corpus is copied out of this cache on every rerun; the label
287 # lets `load_secondary_dataset` key it without hashing it each time.
288 words, fixations = stamp_source((data[0], data[1]))
289 payloads = {
290 table: names.to_payload()
291 for table, names in getattr(data, "column_names", {}).items()
292 }
293 return words, fixations, payloads
296@st.cache_data(show_spinner=False) # UX-168: B's dataset card covers this.
297def _load_builtin_frames(name: str) -> tuple[pd.DataFrame, pd.DataFrame]:
298 """Normalized frames for the bundled demo / synthetic trial.
300 Both are small and packaged, so caching the *normalized* result here is the
301 whole cost — the raw loaders they call are already cached themselves.
302 Normalizing reports nothing on this path, so it reports once, first thing:
303 only a miss gets here, and B's gated card waits for a report (UX-166).
304 """
305 from scanpath_studio import api
306 from scanpath_studio.data import load_sample_data
307 from scanpath_studio.synthetic import load_synthetic_data
309 progress.report()
310 raw = load_sample_data() if name == DEMO_CHOICE else load_synthetic_data()
311 words, fixations = api.load_scanpath_data(raw[0], raw[1], names="canonical")
312 return words, fixations
315@st.cache_data(show_spinner=False)
316def _builtin_column_names(name: str) -> dict[str, dict]:
317 """DATA-66: the demo's / synthetic trial's own column names, for a B drawn
318 from them — read from the same raw frames and auto-detected schemas
319 `_load_builtin_frames` normalizes, as payloads — with the columns that
320 load rewrites marked (the demo's word ids are shifted onto its boxes)."""
321 from scanpath_studio.column_names import for_tables
322 from scanpath_studio.data import (
323 harmonize_frames_reporting,
324 load_sample_data,
325 normalize_fixations,
326 normalize_words,
327 propose_fix_schema,
328 propose_word_schema,
329 )
330 from scanpath_studio.synthetic import load_synthetic_data
332 words, fixations = (
333 load_sample_data() if name == DEMO_CHOICE else load_synthetic_data()
334 )
335 schemas = {
336 "words": propose_word_schema(words),
337 "fixations": propose_fix_schema(fixations),
338 }
339 *_frames, rewrites = harmonize_frames_reporting(
340 normalize_words(words, schemas["words"]),
341 normalize_fixations(fixations, schemas["fixations"]),
342 )
343 return for_tables(
344 schemas, {"words": words, "fixations": fixations}, rewrites=rewrites
345 )
348def source_has_raw_gaze(name: str | None) -> bool:
349 """Whether comparison source ``name`` carries raw gaze, without loading it.
351 The rail is drawn before B's dataset loads, and its 🔵 Raw gaze switch must
352 be live when only B's dataset has samples (VIZ-48). A stored upload says so
353 in its frame, the demo always has some, the public corpora never do.
354 """
355 if not name or name == THIS_DATASET:
356 return False
357 stored = (st.session_state.get("_datasets") or {}).get(name)
358 if isinstance(stored, dict):
359 raw = stored.get("raw_gaze")
360 return raw is not None and not raw.empty
361 return name == DEMO_CHOICE
364@st.cache_data(show_spinner=False)
365def _load_demo_raw_gaze() -> pd.DataFrame:
366 """The bundled demo's normalized raw gaze, for a demo B (VIZ-48)."""
367 from scanpath_studio import api
369 return api.load_sample_raw_gaze(names="canonical")
372def snapshot_for(
373 name: str, words: pd.DataFrame, fixations: pd.DataFrame
374) -> SetupSnapshot:
375 """One named source's own screen — never the live ``global_*`` keys.
377 A stored upload carries the snapshot its wizard captured. Anything else goes
378 through the one source→monitor table (`app.resolve_source_monitor`): a corpus
379 that declares a presentation monitor reports ``MEASURED``, and a corpus whose
380 canvas is inferred from data extents reports ``ESTIMATED``. Physical size,
381 viewing distance and typography stay at their defaults, marked ``ASSUMED`` —
382 no registry entry records them, and B's panel does not use them.
384 **Name-driven, not B-specific** (CMP-11). The `global_*` keys describe
385 whichever dataset is *active*, so resolving B through here and A through
386 `app.active_setup_snapshot` would report different provenance for the same
387 corpus depending on which side of a comparison it landed on — and
388 `experimental_setup.setups_comparable` gates on provenance, so that
389 asymmetry would make the overlay legal one way round and illegal the other.
390 Both sides go through this function.
391 """
392 from scanpath_studio import app
394 stored = (st.session_state.get("_datasets") or {}).get(name)
395 if isinstance(stored, dict) and isinstance(stored.get("setup"), dict):
396 return SetupSnapshot.from_dict(stored["setup"], fallback=SetupSnapshot())
397 # A built-in or public dataset whose setup the user saved is that setup.
398 if (override := app.dataset_setup_override(name)) is not None:
399 return override
400 width, height, authoritative = app.resolve_source_monitor(name, words, fixations)
401 return SetupSnapshot(
402 canvas_width=int(width),
403 canvas_height=int(height),
404 screen_provenance=(
405 Provenance.MEASURED if authoritative else Provenance.ESTIMATED
406 ),
407 geometry_provenance=Provenance.ASSUMED,
408 text_provenance=Provenance.ASSUMED,
409 )
412#: Pre-CMP-11 private name, kept so existing call sites and tests resolve.
413_snapshot_for = snapshot_for
416def load_secondary_dataset(name: str | None) -> SecondaryDataset | None:
417 """Load one comparison source by name, or ``None`` when there is nothing to load.
419 ``None`` / `THIS_DATASET` / an unready name all return ``None`` — the caller
420 then behaves exactly as it did before CMP-8 (B out of A's own pool).
421 """
422 from scanpath_studio.utils import build_combo_options_for
424 if not name or name == THIS_DATASET:
425 return None
426 stored = (st.session_state.get("_datasets") or {}).get(name)
427 raw_gaze = None
428 column_names: dict[str, ColumnNames] = {}
429 if isinstance(stored, dict):
430 words, fixations = stored["words"], stored["fixations"]
431 composite = tuple(stored.get("composite_trial_columns") or ())
432 raw_gaze = stored.get("raw_gaze")
433 column_names = {
434 table: ColumnNames.from_payload(payload)
435 for table, payload in (stored.get("column_names") or {}).items()
436 }
437 vouch_for_frames((words, fixations, raw_gaze))
438 else:
439 from scanpath_studio import app
441 if name in app.public_dataset_registry():
442 # Re-check readiness directly rather than rebuilding the whole option
443 # list: that would re-sweep *every* corpus' filesystem a second time
444 # on the very rerun a cross-dataset pick already costs the most.
445 if not _public_ready(name)[0]:
446 return None
447 root, options = _public_location(name)
448 words, fixations, payloads = _load_public_frames(
449 name, root, tuple(sorted(options.items()))
450 )
451 adopt_source(words, fixations)
452 column_names = {
453 table: ColumnNames.from_payload(payload)
454 for table, payload in payloads.items()
455 }
456 elif name in (DEMO_CHOICE, SYNTHETIC_CHOICE):
457 words, fixations = _load_builtin_frames(name)
458 column_names = {
459 table: ColumnNames.from_payload(payload)
460 for table, payload in _builtin_column_names(name).items()
461 }
462 if name == DEMO_CHOICE:
463 raw_gaze = _load_demo_raw_gaze()
464 else:
465 return None
466 composite = ()
467 if fixations is None or fixations.empty:
468 return None
469 combos, _, _ = build_combo_options_for(fixations, composite)
470 return SecondaryDataset(
471 name=name,
472 words=words,
473 fixations=fixations,
474 combos=combos,
475 setup=snapshot_for(name, words, fixations),
476 composite_trial_columns=composite,
477 raw_gaze=raw_gaze if raw_gaze is not None and not raw_gaze.empty else None,
478 column_names=column_names,
479 )