Coverage for scanpath_studio/compare_source.py: 89%

169 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""The *second* dataset a comparison can draw scanpath B from (CMP-8 §2). 

2 

3Compare mode used to pick B out of the same loaded corpus as A. This module is 

4what lets it reach a different one — a PoTeC reader beside a OneStop reader, or 

5the same text read under two corpora. 

6 

7**The one hard constraint: nothing here may render.** The app's public-corpus 

8loaders (`app._load_public_dataset`) draw directory inputs, *Expected files* 

9layouts and ⬇ Download buttons; none of that can appear inside the compare 

10picker, which is a selectbox in the middle of the plot column. So this module 

11reads the *location state those loaders already wrote* (`<prefix>_dir`) and goes 

12straight to the widget-free `datasets.load_*` functions. A corpus whose location 

13has never been set is offered **disabled** with a reason, never loaded blind. 

14 

15`app` is imported lazily inside the functions, not at module scope: `app` 

16imports `tabs`, `tabs` imports this, so a module-level import would close the 

17cycle (the same reason `wizard.py` is imported lazily by `app`). 

18""" 

19 

20from __future__ import annotations 

21 

22from collections.abc import Mapping 

23from dataclasses import dataclass, field 

24 

25import pandas as pd 

26import streamlit as st 

27 

28from . import progress 

29from .column_names import ColumnNames 

30from .constants import ( 

31 DEMO_CHOICE, 

32 EYEGENBENCH_DEFAULT_DIR, 

33 MULTIPLEYE_DEFAULT_DIR, 

34 ONESTOP_PUBLIC_DEFAULT_DIR, 

35 POTEC_DEFAULT_DIR, 

36 SYNTHETIC_CHOICE, 

37 onestop_regime_for_choice, 

38) 

39from .data import adopt_source, stamp_source, vouch_for_frames 

40from .experimental_setup import Provenance, SetupSnapshot 

41from .session_keys import COMPARE_SOURCE_STATE_KEY 

42 

43#: The picker's "stay in this dataset" entry — compare mode's behaviour before 

44#: CMP-8, and still the default. 

45THIS_DATASET = "This dataset" 

46 

47#: Session key holding the picked secondary source name. Re-exported from 

48#: `session_keys` rather than spelled again: it is a deep-link-seeded key, and 

49#: two literals for one wire-format name is exactly the drift `session_keys.py` 

50#: exists to prevent. 

51COMPARE_SOURCE_KEY = COMPARE_SOURCE_STATE_KEY 

52 

53#: `<key_prefix>_dir` session keys written by `app._dataset_dir_input`, plus the 

54#: default each loader passes it. Read-only here: this module never writes a 

55#: location, it only reports whether one is usable. 

56_MULTIPLEYE_LABEL_HINT = "MultiplEYE" 

57_POTEC_LABEL_HINT = "PoTeC" 

58 

59 

60@dataclass(frozen=True, eq=False) 

61class SecondaryDataset: 

62 """One loaded comparison source: normalized frames plus what they were shown on. 

63 

64 ``eq=False`` because the frames make dataclass equality ambiguous (pandas 

65 raises on a truth-valued comparison), and nothing compares these. 

66 """ 

67 

68 name: str 

69 words: pd.DataFrame 

70 fixations: pd.DataFrame 

71 combos: pd.DataFrame 

72 setup: SetupSnapshot 

73 composite_trial_columns: tuple[str, ...] = field(default=()) 

74 #: VIZ-48: the source's normalized raw gaze, when it carries any — an 

75 #: upload's, or the bundled demo's. The public corpora ship none. 

76 raw_gaze: pd.DataFrame | None = None 

77 #: DATA-66: B's column-name map per table — a stored upload's own; empty for 

78 #: a public corpus or the demo until their loaders return one (phase 4). 

79 column_names: Mapping[str, ColumnNames] = field(default_factory=dict) 

80 

81 

82def _resolved_dir(key: str, default_dir: str) -> str: 

83 """The directory a public-corpus loader would read, without rendering it. 

84 

85 Mirrors `app._dataset_dir_input`'s return value: the session key the text 

86 input wrote, else the loader's default, resolved against the project root. 

87 """ 

88 from scanpath_studio import app 

89 

90 raw = str(st.session_state.get(key) or "").strip() or default_dir 

91 if app.data_root() and not app.local_filesystem_enabled(): 

92 # S2: on a shared deployment the path box isn't rendered at all and the 

93 # location comes from the server's environment — the same rule the 

94 # loader itself follows. 

95 return str(app.data_root()) 

96 return app._resolve_data_dir(raw) 

97 

98 

99def _public_location(label: str) -> tuple[str, dict]: 

100 """``(root, loader kwargs)`` for a `PUBLIC_DATASET_REGISTRY` label. 

101 

102 The kwargs are the *user's current* source options (OneStop's variant / 

103 regime / parts, MultiplEYE's fixation source) — the same session keys the 

104 the Compare-with widgets own, read rather than re-rendered. 

105 """ 

106 from scanpath_studio import app, datasets 

107 

108 # DATA-27 (Task 11R): a prepared benchmark corpus is one registry entry that 

109 # names the corpus inside the bundle, so it dispatches on `benchmark_dataset` 

110 # — **before** the label-substring branches below, which a harmonised 

111 # "PoTeC …" / "OneStop …" label would otherwise match and send to the native 

112 # loader. 

113 spec = app.public_dataset_registry().get(label) or {} 

114 if dataset := spec.get("benchmark_dataset"): 

115 return _resolved_dir("eyegenbench_dir", EYEGENBENCH_DEFAULT_DIR), { 

116 "dataset": dataset, 

117 } 

118 if regime := onestop_regime_for_choice(label): 

119 # DATA-63: one dataset per regime, every part, from the public release. 

120 # UX-184: the box's own default — under the Download folder. 

121 default = app._download_target(ONESTOP_PUBLIC_DEFAULT_DIR) 

122 return _resolved_dir("onestop_public_dir", default), { 

123 "variant": "public", 

124 "regime": regime, 

125 "parts": tuple(datasets.onestop_regime_parts(regime)), 

126 } 

127 if _POTEC_LABEL_HINT in label: 

128 return _resolved_dir("potec_dir", app._download_target(POTEC_DEFAULT_DIR)), {} 

129 if _MULTIPLEYE_LABEL_HINT in label: 

130 return _resolved_dir("multipleye_dir", MULTIPLEYE_DEFAULT_DIR), { 

131 "fixation_source": str( 

132 st.session_state.get("multipleye_fixation_source") or "scanpaths" 

133 ), 

134 } 

135 return "", {} 

136 

137 

138def _public_ready(label: str) -> tuple[bool, str]: 

139 """Whether a public corpus can be loaded *silently*, and why not if it can't. 

140 

141 Uses the existing readiness helpers — `datasets.potec_present` / 

142 `onestop_present` / `multipleye_inventory` — so a corpus is never offered as 

143 B unless the very same check the main source picker runs says its files are 

144 there. A download is deliberately never triggered from here. 

145 

146 Cached on the resolved location + source options: this runs for *every* 

147 registry entry on every rerun that Compare is on, and `multipleye_inventory` 

148 walks each session directory (`app._cached_multipleye_inventory` wraps the 

149 same call for the same reason). Location changes bust the key. 

150 """ 

151 root, kwargs = _public_location(label) 

152 return _public_ready_cached( 

153 label, root, tuple(sorted(kwargs.items())), _short_name(label) 

154 ) 

155 

156 

157def _short_name(label: str) -> str: 

158 """What to call this corpus in a disabled entry's hint. 

159 

160 The registry's own ``short`` name, not the label's prefix: two entries can 

161 share one prefix — the native and the harmonised PoTeC both split to 

162 ``"PoTeC"`` — so a prefix-derived hint told the user to go and open one of 

163 two entries it couldn't tell apart (M12). 

164 """ 

165 from scanpath_studio import app 

166 

167 spec = app.public_dataset_registry().get(label) or {} 

168 return str(spec.get("short") or "").strip() or label.split(" — ")[0] 

169 

170 

171@st.cache_data(show_spinner=False) 

172def _public_ready_cached( 

173 label: str, root: str, options: tuple, short: str = "" 

174) -> tuple[bool, str]: 

175 from scanpath_studio import datasets 

176 

177 kwargs = dict(options) 

178 short = short or label.split(" — ")[0] 

179 if not root: 

180 return False, f"{short} has no data folder yet — open it once first." 

181 hint = f"Open {short} as the main dataset once to download or locate it." 

182 try: 

183 if dataset := kwargs.get("dataset"): 

184 from scanpath_studio.eyegenbench import eyegenbench_present 

185 

186 present = eyegenbench_present(root, dataset) 

187 elif onestop_regime_for_choice(label): 

188 present = datasets.onestop_present( 

189 root, 

190 regime=kwargs["regime"], 

191 parts=list(kwargs["parts"]), 

192 variant=kwargs["variant"], 

193 ) 

194 elif _POTEC_LABEL_HINT in label: 

195 present = datasets.potec_present(root) 

196 elif _MULTIPLEYE_LABEL_HINT in label: 

197 sessions, _ = datasets.multipleye_inventory( 

198 root, fixation_source=kwargs["fixation_source"] 

199 ) 

200 present = bool(sessions) 

201 else: 

202 return False, f"{short} isn't loadable as a comparison dataset." 

203 except (OSError, ValueError, KeyError): 

204 # `KeyError` because a manifest row is data from a file on disk and can 

205 # be malformed — a row with no `name` used to escape this catch and take 

206 # the whole app down through the compare-B enumeration, which runs over 

207 # *every* registry entry on every rerun Compare is on (I2). The nameless 

208 # row is now skipped at the source too (`eyegenbench.entry_name`); this 

209 # is the belt to that braces, since the same catch covers four loaders' 

210 # readiness probes and only one of them has been hardened. 

211 present = False 

212 return (True, "") if present else (False, hint) 

213 

214 

215def secondary_dataset_options( 

216 *, exclude: str | None = None 

217) -> list[tuple[str, bool, str]]: 

218 """Every source compare mode could draw B from: ``(name, ready, why_not)``. 

219 

220 Stored uploads, the bundled demo and the synthetic trial are always ready — 

221 they are in memory or in the package. Public corpora are *always offered* but 

222 ready only when their files are already where the main picker last looked; 

223 an unready entry renders disabled with ``why_not`` rather than disappearing, 

224 so the capability is discoverable instead of mysteriously absent. 

225 

226 The ``$ONESTOP_DATA_DIR`` **server bundle is deliberately not offered.** 

227 `data.load_onestop_server_bundle` is sub-second only when it is given a 

228 participant to load a per-pid shard for; without one it falls back to the 

229 full CSV exports — its own docstring says ~3 min and ~60 GB for the L2 

230 cohort. The picker has no participant to give at the moment it builds its 

231 options, so offering the bundle would mean blocking the whole app for 

232 minutes, and possibly OOM-ing the server, to draw one comparison trial. The 

233 same corpus is reachable as the public *OneStop* entry below. 

234 

235 ``exclude`` drops one name (the active source — comparing a dataset with 

236 itself is what `THIS_DATASET` already means). 

237 """ 

238 from scanpath_studio import app 

239 

240 options: list[tuple[str, bool, str]] = [ 

241 (name, True, "") for name in sorted(st.session_state.get("_datasets") or {}) 

242 ] 

243 options.append((DEMO_CHOICE, True, "")) 

244 options.append((SYNTHETIC_CHOICE, True, "")) 

245 if app.public_datasets_enabled(): 

246 options.extend( 

247 (label, *_public_ready(label)) for label in app.public_dataset_registry() 

248 ) 

249 return [option for option in options if option[0] != exclude] 

250 

251 

252@st.cache_data(show_spinner=False) # UX-168: B's dataset card covers this. 

253def _load_public_frames( 

254 label: str, root: str, options: tuple 

255) -> tuple[pd.DataFrame, pd.DataFrame, dict]: 

256 """Normalized frames for a public corpus, keyed on its location + options, 

257 and its column-name map per table, as payloads (DATA-66). 

258 

259 Goes through the `datasets.load_*` entry points (which normalize internally 

260 via `api.load_scanpath_data`), never `app.prepare_data` — that one takes a 

261 ``mapping_host`` and renders the column-mapping panels. 

262 """ 

263 from scanpath_studio import datasets 

264 

265 kwargs = dict(options) 

266 # The app works in the internal names; DATA-66: the corpus' own names come 

267 # back beside the frames (`ScanpathData.column_names`), for B's labels. 

268 if dataset := kwargs.get("dataset"): 

269 from scanpath_studio.eyegenbench import load_eyegenbench 

270 

271 data = load_eyegenbench(root, dataset=dataset, names="canonical") 

272 elif onestop_regime_for_choice(label): 

273 data = datasets.load_onestop( 

274 root, 

275 regime=kwargs["regime"], 

276 parts=list(kwargs["parts"]), 

277 variant=kwargs["variant"], 

278 names="canonical", 

279 ) 

280 elif _POTEC_LABEL_HINT in label: 

281 data = datasets.load_potec(root, names="canonical") 

282 else: 

283 data = datasets.load_multipleye( 

284 root, fixation_source=kwargs["fixation_source"], names="canonical" 

285 ) 

286 # BUG-103: B's corpus is copied out of this cache on every rerun; the label 

287 # lets `load_secondary_dataset` key it without hashing it each time. 

288 words, fixations = stamp_source((data[0], data[1])) 

289 payloads = { 

290 table: names.to_payload() 

291 for table, names in getattr(data, "column_names", {}).items() 

292 } 

293 return words, fixations, payloads 

294 

295 

296@st.cache_data(show_spinner=False) # UX-168: B's dataset card covers this. 

297def _load_builtin_frames(name: str) -> tuple[pd.DataFrame, pd.DataFrame]: 

298 """Normalized frames for the bundled demo / synthetic trial. 

299 

300 Both are small and packaged, so caching the *normalized* result here is the 

301 whole cost — the raw loaders they call are already cached themselves. 

302 Normalizing reports nothing on this path, so it reports once, first thing: 

303 only a miss gets here, and B's gated card waits for a report (UX-166). 

304 """ 

305 from scanpath_studio import api 

306 from scanpath_studio.data import load_sample_data 

307 from scanpath_studio.synthetic import load_synthetic_data 

308 

309 progress.report() 

310 raw = load_sample_data() if name == DEMO_CHOICE else load_synthetic_data() 

311 words, fixations = api.load_scanpath_data(raw[0], raw[1], names="canonical") 

312 return words, fixations 

313 

314 

315@st.cache_data(show_spinner=False) 

316def _builtin_column_names(name: str) -> dict[str, dict]: 

317 """DATA-66: the demo's / synthetic trial's own column names, for a B drawn 

318 from them — read from the same raw frames and auto-detected schemas 

319 `_load_builtin_frames` normalizes, as payloads — with the columns that 

320 load rewrites marked (the demo's word ids are shifted onto its boxes).""" 

321 from scanpath_studio.column_names import for_tables 

322 from scanpath_studio.data import ( 

323 harmonize_frames_reporting, 

324 load_sample_data, 

325 normalize_fixations, 

326 normalize_words, 

327 propose_fix_schema, 

328 propose_word_schema, 

329 ) 

330 from scanpath_studio.synthetic import load_synthetic_data 

331 

332 words, fixations = ( 

333 load_sample_data() if name == DEMO_CHOICE else load_synthetic_data() 

334 ) 

335 schemas = { 

336 "words": propose_word_schema(words), 

337 "fixations": propose_fix_schema(fixations), 

338 } 

339 *_frames, rewrites = harmonize_frames_reporting( 

340 normalize_words(words, schemas["words"]), 

341 normalize_fixations(fixations, schemas["fixations"]), 

342 ) 

343 return for_tables( 

344 schemas, {"words": words, "fixations": fixations}, rewrites=rewrites 

345 ) 

346 

347 

348def source_has_raw_gaze(name: str | None) -> bool: 

349 """Whether comparison source ``name`` carries raw gaze, without loading it. 

350 

351 The rail is drawn before B's dataset loads, and its 🔵 Raw gaze switch must 

352 be live when only B's dataset has samples (VIZ-48). A stored upload says so 

353 in its frame, the demo always has some, the public corpora never do. 

354 """ 

355 if not name or name == THIS_DATASET: 

356 return False 

357 stored = (st.session_state.get("_datasets") or {}).get(name) 

358 if isinstance(stored, dict): 

359 raw = stored.get("raw_gaze") 

360 return raw is not None and not raw.empty 

361 return name == DEMO_CHOICE 

362 

363 

364@st.cache_data(show_spinner=False) 

365def _load_demo_raw_gaze() -> pd.DataFrame: 

366 """The bundled demo's normalized raw gaze, for a demo B (VIZ-48).""" 

367 from scanpath_studio import api 

368 

369 return api.load_sample_raw_gaze(names="canonical") 

370 

371 

372def snapshot_for( 

373 name: str, words: pd.DataFrame, fixations: pd.DataFrame 

374) -> SetupSnapshot: 

375 """One named source's own screen — never the live ``global_*`` keys. 

376 

377 A stored upload carries the snapshot its wizard captured. Anything else goes 

378 through the one source→monitor table (`app.resolve_source_monitor`): a corpus 

379 that declares a presentation monitor reports ``MEASURED``, and a corpus whose 

380 canvas is inferred from data extents reports ``ESTIMATED``. Physical size, 

381 viewing distance and typography stay at their defaults, marked ``ASSUMED`` — 

382 no registry entry records them, and B's panel does not use them. 

383 

384 **Name-driven, not B-specific** (CMP-11). The `global_*` keys describe 

385 whichever dataset is *active*, so resolving B through here and A through 

386 `app.active_setup_snapshot` would report different provenance for the same 

387 corpus depending on which side of a comparison it landed on — and 

388 `experimental_setup.setups_comparable` gates on provenance, so that 

389 asymmetry would make the overlay legal one way round and illegal the other. 

390 Both sides go through this function. 

391 """ 

392 from scanpath_studio import app 

393 

394 stored = (st.session_state.get("_datasets") or {}).get(name) 

395 if isinstance(stored, dict) and isinstance(stored.get("setup"), dict): 

396 return SetupSnapshot.from_dict(stored["setup"], fallback=SetupSnapshot()) 

397 # A built-in or public dataset whose setup the user saved is that setup. 

398 if (override := app.dataset_setup_override(name)) is not None: 

399 return override 

400 width, height, authoritative = app.resolve_source_monitor(name, words, fixations) 

401 return SetupSnapshot( 

402 canvas_width=int(width), 

403 canvas_height=int(height), 

404 screen_provenance=( 

405 Provenance.MEASURED if authoritative else Provenance.ESTIMATED 

406 ), 

407 geometry_provenance=Provenance.ASSUMED, 

408 text_provenance=Provenance.ASSUMED, 

409 ) 

410 

411 

412#: Pre-CMP-11 private name, kept so existing call sites and tests resolve. 

413_snapshot_for = snapshot_for 

414 

415 

416def load_secondary_dataset(name: str | None) -> SecondaryDataset | None: 

417 """Load one comparison source by name, or ``None`` when there is nothing to load. 

418 

419 ``None`` / `THIS_DATASET` / an unready name all return ``None`` — the caller 

420 then behaves exactly as it did before CMP-8 (B out of A's own pool). 

421 """ 

422 from scanpath_studio.utils import build_combo_options_for 

423 

424 if not name or name == THIS_DATASET: 

425 return None 

426 stored = (st.session_state.get("_datasets") or {}).get(name) 

427 raw_gaze = None 

428 column_names: dict[str, ColumnNames] = {} 

429 if isinstance(stored, dict): 

430 words, fixations = stored["words"], stored["fixations"] 

431 composite = tuple(stored.get("composite_trial_columns") or ()) 

432 raw_gaze = stored.get("raw_gaze") 

433 column_names = { 

434 table: ColumnNames.from_payload(payload) 

435 for table, payload in (stored.get("column_names") or {}).items() 

436 } 

437 vouch_for_frames((words, fixations, raw_gaze)) 

438 else: 

439 from scanpath_studio import app 

440 

441 if name in app.public_dataset_registry(): 

442 # Re-check readiness directly rather than rebuilding the whole option 

443 # list: that would re-sweep *every* corpus' filesystem a second time 

444 # on the very rerun a cross-dataset pick already costs the most. 

445 if not _public_ready(name)[0]: 

446 return None 

447 root, options = _public_location(name) 

448 words, fixations, payloads = _load_public_frames( 

449 name, root, tuple(sorted(options.items())) 

450 ) 

451 adopt_source(words, fixations) 

452 column_names = { 

453 table: ColumnNames.from_payload(payload) 

454 for table, payload in payloads.items() 

455 } 

456 elif name in (DEMO_CHOICE, SYNTHETIC_CHOICE): 

457 words, fixations = _load_builtin_frames(name) 

458 column_names = { 

459 table: ColumnNames.from_payload(payload) 

460 for table, payload in _builtin_column_names(name).items() 

461 } 

462 if name == DEMO_CHOICE: 

463 raw_gaze = _load_demo_raw_gaze() 

464 else: 

465 return None 

466 composite = () 

467 if fixations is None or fixations.empty: 

468 return None 

469 combos, _, _ = build_combo_options_for(fixations, composite) 

470 return SecondaryDataset( 

471 name=name, 

472 words=words, 

473 fixations=fixations, 

474 combos=combos, 

475 setup=snapshot_for(name, words, fixations), 

476 composite_trial_columns=composite, 

477 raw_gaze=raw_gaze if raw_gaze is not None and not raw_gaze.empty else None, 

478 column_names=column_names, 

479 )