Coverage for scanpath_studio/wizard.py: 86%

1168 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""The Upload / Add-dataset wizard (the main-area guided data-setup flow). 

2 

3Split out of ``app.py``: everything from the upload-table reading UI through the 

4ordered wizard steps (identity → trial/participant/text → keep-fields/filters → 

5name & finish) and the MultiplEYE upload branch. ``app.py`` drives this from 

6``resolve_data_source`` / ``main`` and re-exports a few helpers for tests. 

7 

8A handful of data-IO / normalization helpers (``_read_uploaded_frame``, 

9``_normalize_pair``, ``_stash_active_mapping``, ``_render_unmapped_view``) stay in 

10``app.py`` and are reached via ``app.<name>`` at call time — that keeps the 

11``app._read_uploaded_frame`` upload seam (monkeypatched in AppTests) intact and 

12avoids an app⇄wizard import cycle. 

13""" 

14 

15from __future__ import annotations 

16 

17import json 

18import re 

19from typing import NamedTuple 

20 

21import pandas as pd 

22import streamlit as st 

23 

24from . import app, wizard_shell 

25from .column_names import ColumnNames, for_tables 

26from .constants import ( 

27 _VIEW_DATA, 

28 CITATION, 

29 DATASET_ADDED_KEY, 

30 DATASET_DESCRIPTIONS_KEY, 

31 DEMO_CHOICE, 

32 FONT_FAMILY, 

33 ICONS, 

34 TRIAL_IDENTITY_CHECK_KEY, 

35 WIZARD_LEAVE_KEY, 

36 multipleye_upload_enabled, 

37 plural, 

38 upload_identity, 

39 upload_limit_label, 

40 upload_limit_mb, 

41) 

42from .controls import ( 

43 _GRID_LABEL_W, 

44 ADD_ATTEMPTED_KEY, 

45 FIX_FIELD_SPECS, 

46 RAW_GAZE_FIELD_SPECS, 

47 TOUCHED_FIELDS_KEY, 

48 WORD_FIELD_SPECS, 

49 _mark_field_touched, 

50 claim_mapping, 

51 column_mapping_ui, 

52 inline_field_label, 

53 mark_cells, 

54 multi_field_flag, 

55 value_preview_tip, 

56) 

57from .crash_report import guarded 

58from .data import ( 

59 FIX_OPTIONAL_FIELDS, 

60 PARTICIPANT_CANDIDATES, 

61 READING_MEASURE_KEYS, 

62 SOURCE_FILE_COLUMN, 

63 WORD_OPTIONAL_FIELDS, 

64 aggregate_char_boxes, 

65 assign_derived, 

66 canvas_geometry_frames, 

67 categorize_columns, 

68 compute_canvas_size, 

69 compute_keep_columns, 

70 dropped_columns, 

71 empty_fixations_frame, 

72 empty_words_frame, 

73 extract_columns_from_source_file, 

74 frame_cache, 

75 frame_fingerprint, 

76 looks_like_condition, 

77 normalization_issues, 

78 normalize_raw_gaze, 

79 pick_column, 

80 propose_fix_schema, 

81 propose_raw_gaze_schema, 

82 propose_word_schema, 

83 source_file_regex_collisions, 

84 split_source_file, 

85 trial_id_series, 

86 trial_keys, 

87 trial_mapping_columns, 

88 validate_fix_schema, 

89 validate_raw_gaze_schema, 

90 validate_word_schema, 

91) 

92from .experimental_setup import ( 

93 SETUP_GROUP_LABELS, 

94 SETUP_GROUPS, 

95 Provenance, 

96 SetupSnapshot, 

97 font_pt_to_px, 

98) 

99from .menu import view_label 

100from .persistence import is_loopback_url, rename_cached_dataset 

101from .session_keys import COMPARE_SOURCE_STATE_KEY 

102from .styles import mapping_menu_css 

103from .synthetic import EXAMPLE_ZIP_FILE, example_import_zip 

104from .tabs import _collect_column_mapping 

105from .tour import ( 

106 maybe_show_wizard_guide, 

107 render_spotlight_wizard_guide, 

108 render_wizard_guide_button, 

109) 

110from .url_state import PLOT_CONFIG_SCHEMA, _seed_column_mapping 

111 

112 

113class _UploadResult(NamedTuple): 

114 """Result of the grouped-upload flow. 

115 

116 ``words``/``fixations``/``raw_gaze`` are normalized (empty when absent or, for 

117 words/fixations, when the mapping is incomplete). ``raw_words``/``raw_fixations`` 

118 are the pre-normalization frames shown by ``app._render_unmapped_view`` when 

119 ``problems`` is non-empty.""" 

120 

121 words: pd.DataFrame 

122 fixations: pd.DataFrame 

123 raw_gaze: pd.DataFrame 

124 raw_words: pd.DataFrame 

125 raw_fixations: pd.DataFrame 

126 problems: list 

127 

128 

129#: BUG-32: the dataset the wizard's `col_map_*` mapping describes. The 🗂️ Data 

130#: page maps a built-in source under the same keys, keyed by its source, so 

131#: this identity is what tells the two apart when the headers match: a pick 

132#: made here never carries back into the demo or a public corpus, and theirs 

133#: never into a new upload. One constant serves every add-dataset session, 

134#: because entering the wizard resets its mapping anyway. 

135WIZARD_MAPPING_DATASET = "add-dataset wizard" 

136_WIZARD_MAPPING_PREFIXES = ("col_map_words", "col_map_fix", "col_map_raw_gaze") 

137 

138 

139def _reset_wizard_widgets() -> None: 

140 """Clear the wizard's per-table mapping + keep-field widgets so 'Add data' 

141 starts a fresh dataset.""" 

142 # BUG-32: from here these keys describe the dataset being added. Its first 

143 # upload keeps a setup restored before it; ✕ Cancel leaves nothing the demo 

144 # would adopt as its own. 

145 for prefix in _WIZARD_MAPPING_PREFIXES: 

146 claim_mapping(prefix, WIZARD_MAPPING_DATASET) 

147 for key in [ 

148 k 

149 for k in list(st.session_state.keys()) 

150 # UX-114: the per-table keep pickers (`wizard_keep_col_map_words` etc., 

151 # plus their Select-all/None button keys) share the "col_map_" table 

152 # names as a *suffix*, not a prefix — sweep "wizard_keep_" too, or a 

153 # new dataset would inherit the previous one's keep choices. 

154 if isinstance(k, str) and k.startswith(("col_map_", "wizard_keep_")) 

155 ]: 

156 del st.session_state[key] 

157 for key in ( 

158 "wizard_dataset_name", 

159 "wizard_dataset_description", 

160 "wizard_dataset_format", 

161 "wizard_config_restore", 

162 "_wizard_config_last", 

163 "_wizard_restored_meta", 

164 "_composite_trial_columns", 

165 "wizard_filter_fields", 

166 # MultiplEYE preset uploads + generic filename-derivation / aggregation. 

167 "mpe_fix_upload", 

168 "mpe_aoi_upload", 

169 "mpe_questions_upload", 

170 "mpe_participant_upload", 

171 "wizard_filename_split", 

172 "wizard_filename_mode", 

173 "wizard_filename_regex", 

174 "wizard_filename_regex_lower", 

175 _FILENAME_DERIVE_LINE_COUNT_KEY, 

176 "wizard_aggregate_char_boxes", 

177 # UX-53: a new dataset starts unattempted, so its required fields are 

178 # blank rather than red until this one is asked to be added — and 

179 # unconfirmed, so nothing is green until somebody picks it here. 

180 ADD_ATTEMPTED_KEY, 

181 TOUCHED_FIELDS_KEY, 

182 # DATA-22 Recording-setup step. The *mode* radios always reset — decision 

183 # (d): a second dataset from the same lab keeps the pre-filled values 

184 # (`_wizard_setup_recall`, deliberately NOT cleared here) but the user 

185 # still has to assert that the setup applies to this dataset too. 

186 *(_SETUP_MODE_KEYS[g] for g in SETUP_GROUPS), 

187 "wizard_setup_screen_w", 

188 "wizard_setup_screen_h", 

189 "wizard_setup_monitor_mm", 

190 "wizard_setup_distance_mm", 

191 "wizard_setup_font_pt", 

192 "wizard_setup_font_family", 

193 "_wizard_restored_setup", 

194 "_wizard_setup_restored_applied", 

195 "_wizard_problems_last", 

196 ): 

197 st.session_state.pop(key, None) 

198 # Clear the steps' open flags so the next entry does not open wherever the 

199 # previous dataset was left. 

200 wizard_shell.reset_accordion() 

201 

202 

203def _default_dataset_name() -> str: 

204 """A unique 'Dataset N' name not already taken by a stored dataset.""" 

205 existing = st.session_state.get("_datasets", {}) 

206 n = len(existing) + 1 

207 while f"Dataset {n}" in existing: 

208 n += 1 

209 return f"Dataset {n}" 

210 

211 

212def _safe_dataset_name(name: str | None, *, exclude: str | None = None) -> str: 

213 """A non-empty dataset name that collides with neither a built-in source label 

214 nor an already-stored dataset (suffixed ``(2)``, ``(3)``… rather than silently 

215 overwriting an existing entry's frames). 

216 

217 Built-in labels are every token the picker can list 

218 (``app.reserved_source_names`` — the demo, the samples, the server bundles 

219 and every public corpus): a user dataset shadowing one gets a duplicate 

220 option and hijacks the built-in's load branch. DATA-48: a name that still 

221 holds another dataset's annotation store counts as taken too, so a new or 

222 renamed dataset never inherits annotations that are not its own. 

223 

224 ``exclude`` drops one stored name from the collision check — a rename (DATA-23) 

225 must not read the dataset being renamed as a clash with itself and turn 

226 "My corpus" into "My corpus (2)" on a capitalisation fix.""" 

227 import scanpath_studio.annotations as _annotations 

228 

229 name = (name or "").strip() or _default_dataset_name() 

230 if name in app.reserved_source_names(): 

231 name = f"{name} (uploaded)" 

232 existing = { 

233 key: value 

234 for key, value in st.session_state.get("_datasets", {}).items() 

235 if key != exclude 

236 } 

237 existing.update( 

238 dict.fromkeys(_annotations.dataset_names(st.session_state) - {exclude} - {None}) 

239 ) 

240 if name in existing: 

241 base, n = name, 2 

242 while f"{base} ({n})" in existing: 

243 n += 1 

244 name = f"{base} ({n})" 

245 return name 

246 

247 

248def _wizard_finalize_metadata_pools(payload: dict) -> tuple: 

249 """The just-finalized dataset's own participant/trial/text pools. 

250 

251 UX-116: unlike ``_wizard_reader_ids``/``_wizard_trial_combos``/ 

252 ``_wizard_text_ids`` (which read the *raw*, pre-normalization frames, 

253 because those run live while mapping is still being worked out), this 

254 reads the *normalized* ``words``/``fixations`` already in ``payload`` — 

255 the canonical ``participant_id``/``trial_id``/``text_id`` columns are 

256 exactly what the real dataset will carry, so there is no re-deriving to 

257 do. Called once, from ``_finalize_wizard_dataset``. 

258 """ 

259 words = payload.get("words") 

260 fixations = payload.get("fixations") 

261 combo_frames = [ 

262 frame[["participant_id", "trial_id"]] 

263 for frame in (fixations, words) 

264 if frame is not None and not frame.empty and "participant_id" in frame.columns 

265 ] 

266 combos = ( 

267 pd.concat(combo_frames, ignore_index=True).drop_duplicates() 

268 if combo_frames 

269 else pd.DataFrame(columns=["participant_id", "trial_id"]) 

270 ) 

271 participants = sorted(combos["participant_id"].dropna().astype(str).unique()) 

272 text_series = [ 

273 frame["text_id"] 

274 for frame in (fixations, words) 

275 if frame is not None and not frame.empty and "text_id" in frame.columns 

276 ] 

277 texts = ( 

278 sorted(pd.concat(text_series).dropna().astype(str).unique()) 

279 if text_series 

280 else [] 

281 ) 

282 return participants, combos, texts 

283 

284 

285def _source_recipe( 

286 schemas: dict, uploaded_columns: dict, *, aggregated: bool = False 

287) -> dict: 

288 """A new dataset's ``source_recipe``, read by Share → Code 

289 (`code_snippet.upload_source`): each table's mapping, in the uploaded 

290 files' own column names; the steps the loader cannot replay; and the mapped 

291 columns the files did not hold (made from the file names).""" 

292 derived = sorted( 

293 { 

294 str(column) 

295 for table, schema in schemas.items() 

296 if schema 

297 for value in schema.values() 

298 if value 

299 for column in trial_mapping_columns(value) 

300 if str(column) not in uploaded_columns.get(table, set()) 

301 } 

302 ) 

303 return { 

304 "schemas": {t: dict(s) if s else None for t, s in schemas.items()}, 

305 "steps": ["aggregate_char_boxes"] if aggregated else [], 

306 "derived": derived, 

307 } 

308 

309 

310def _finalize_wizard_dataset() -> None: 

311 """Store the wizard's normalized frames as a named dataset and switch to it. 

312 

313 Runs as the "✅ Add dataset" button's ``on_click`` callback. A callback — 

314 not an inline ``if button:`` handler — is required because a real 

315 ``st.file_uploader`` in the wizard can swallow an inline button click (the 

316 click triggers a rerun in which the uploader re-renders and the handler is 

317 never reached), leaving the dataset unstored. The callback fires as part of 

318 the click event, before the rerun, so it always runs. The frames were stashed 

319 in ``_wizard_finalize_payload`` on the render that drew the button. 

320 

321 **Do not carry this rule into an ``st.dialog`` body — it inverts there.** 

322 A dialog body is a fragment, so an ``on_click`` inside one reruns the modal 

323 alone and leaves the page under it untouched; those buttons take their 

324 *return value* plus an explicit ``st.rerun(scope="app")`` instead. See 

325 ``app._delete_confirmation_dialog`` (BUG-36). The two 

326 prescriptions are opposite and both correct; what decides is only whether 

327 the button sits in a dialog/fragment or on an ordinary page beside an 

328 uploader.""" 

329 payload = st.session_state.pop("_wizard_finalize_payload", None) 

330 if payload is None: 

331 return 

332 # UX-116: the metadata tables' own join was deferred (`live_join=False`) 

333 # while this dataset was still being assembled — do it now, against the 

334 # pools this dataset actually ends up with. No UI here: a callback runs 

335 # before the widgets that would host it exist for this run. 

336 from scanpath_studio.tabs import commit_deferred_metadata 

337 

338 participants, combos, texts = _wizard_finalize_metadata_pools(payload) 

339 commit_deferred_metadata(participants, combos, texts) 

340 ds_name = _safe_dataset_name(st.session_state.get("wizard_dataset_name")) 

341 # UX-174 r2 — the sentence the 🗂️ Data page shows under its name. Kept 

342 # beside the store, not in its entry (see `DATASET_DESCRIPTIONS_KEY`). 

343 if description := str( 

344 st.session_state.get("wizard_dataset_description") or "" 

345 ).strip(): 

346 descriptions = dict(st.session_state.get(DATASET_DESCRIPTIONS_KEY) or {}) 

347 descriptions[ds_name] = description 

348 st.session_state[DATASET_DESCRIPTIONS_KEY] = descriptions 

349 store = st.session_state.setdefault("_datasets", {}) 

350 store[ds_name] = payload 

351 # DATA-47: the tables just attached are this dataset's, not a session-wide 

352 # slot — hand them over before the switch, so the next run has nothing to 

353 # swap (and so the dataset this wizard was opened over keeps its own). 

354 # DATA-48 — likewise its annotations (none yet: nothing is annotated in 

355 # the wizard, but the store it opened on belongs to another dataset). 

356 import scanpath_studio.annotations as _annotations 

357 from scanpath_studio import metadata as _metadata 

358 

359 _metadata.adopt_pending_dataset(st.session_state, ds_name) 

360 _annotations.adopt_pending_dataset(st.session_state, ds_name) 

361 # Apply the source switch through the plain pending key that 

362 # resolve_data_source consumes before the radio instantiates, and 

363 # leave the wizard. 

364 st.session_state["_pending_source_choice"] = ds_name 

365 st.session_state["_show_upload_wizard"] = False 

366 st.session_state["setup_complete"] = True 

367 # Flag the transition so main() shows the dataset card at once (UX-166) 

368 # rather than after its usual delay — otherwise the closed wizard is all the 

369 # user sees (stale DOM) for the seconds the heavy first render takes. 

370 st.session_state["_wizard_finalizing"] = True 

371 # …and ask for VAL-7's Trial ID verdict on the dataset this just created. 

372 # It runs as part of the load that is about to happen (`main` already 

373 # computes the report for the open dataset), so pressing ✅ Add dataset pays 

374 # for one sampled scan and hears about a Trial ID that under-specifies while 

375 # the mapping that produced it is still fresh — instead of a page-wide 

376 # banner it will learn to ignore. See `app._trial_identity_alert_dialog`. 

377 st.session_state[TRIAL_IDENTITY_CHECK_KEY] = "add" 

378 # #374 F30: and confirm it, with its counts, once its frames are loaded. 

379 st.session_state[DATASET_ADDED_KEY] = ds_name 

380 # UX-199: on a deployment that saves nothing, the first upload says so. 

381 app.arm_backup_reminder() 

382 

383 

384def upload_annotations(name: str) -> list[dict]: 

385 """The annotations removing upload ``name`` takes with it (BUG-95). 

386 

387 All of its own: since DATA-48 each dataset holds its annotations, so none 

388 is shared with another dataset that happens to reuse its trial ids — which 

389 is why the removal confirmation can say how many will go. 

390 """ 

391 import scanpath_studio.annotations as annotations_mod 

392 

393 return annotations_mod.store_to_records( 

394 annotations_mod.store_for(st.session_state, name) 

395 ) 

396 

397 

398def _remove_dataset(name: str) -> None: 

399 """Remove a previously added dataset (the ✕ button's ``on_click`` callback). 

400 

401 Pops it from the ``_datasets`` store — with its description, its metadata 

402 tables and its annotations (`upload_annotations`) — and, if 

403 it was the selected source, 

404 switches back to the bundled demo via ``_pending_source_choice`` (applied 

405 before the radio re-instantiates, like the wizard finalize/cancel switch). 

406 

407 **UX-87: deleting a dataset drops the computations derived from it.** Every 

408 figure, table and normalization the app cached for this dataset is dead the 

409 moment its frames leave the session — keeping them only holds memory and 

410 leaves stale results one undo-less click away from being attributed to 

411 whatever takes the name next. ``@st.cache_data`` has no per-key eviction, so 

412 this is ``clear_computation_cache``'s blunt instrument: the survivors are 

413 recomputed on the next rerun from frames that are still loaded, which is the 

414 same lossless trade the 🧹 button makes. 

415 """ 

416 # BUG-95 — its annotations go with it, as the confirmation says (DATA-48: 

417 # the dataset's own store, filed under its name). 

418 import scanpath_studio.annotations as annotations_mod 

419 

420 annotations_mod.forget_dataset(st.session_state, name) 

421 store = st.session_state.get("_datasets", {}) 

422 store.pop(name, None) 

423 # UX-174 r2 — and its description, so a new dataset of that name starts 

424 # without one. 

425 descriptions = dict(st.session_state.get(DATASET_DESCRIPTIONS_KEY) or {}) 

426 if descriptions.pop(name, None) is not None: 

427 st.session_state[DATASET_DESCRIPTIONS_KEY] = descriptions 

428 # DATA-47 — its metadata tables go with it. 

429 from scanpath_studio import metadata as _metadata 

430 

431 _metadata.forget_dataset(st.session_state, name) 

432 if st.session_state.get("data_source_choice") == name: 

433 st.session_state["_pending_source_choice"] = DEMO_CHOICE 

434 app.clear_computation_cache() 

435 

436 

437def rename_dataset(old: str, new: str) -> str | None: 

438 """Rename a stored dataset (DATA-23). Returns the name it actually took. 

439 

440 The name is the **key** into the ``_datasets`` store, so a rename is a re-key, 

441 and every other holder of that string has to move with it: the canonical 

442 ``data_source_choice`` (through the pre-widget ``_pending_source_choice`` seam, 

443 like `_remove_dataset`'s fallback — assigning the widget key inline is 

444 unreliable in a browser), the wizard's ``_prev_source`` return address, and 

445 CMP-8's ``cmp_dataset`` when the comparison draws scanpath B from it. The 

446 ENG-26 recovery cache follows in :func:`persistence.rename_cached_dataset`, 

447 which moves the Parquet files rather than re-encoding them. 

448 

449 Returns ``None`` when there is nothing to rename (unknown dataset, or the new 

450 name resolves to the one it already has). The store is rebuilt in order rather 

451 than re-inserted, so the renamed dataset keeps its place in the source picker. 

452 """ 

453 store = st.session_state.get("_datasets", {}) 

454 if old not in store: 

455 return None 

456 name = _safe_dataset_name(new, exclude=old) 

457 if name == old: 

458 return None 

459 # DATA-48: the annotations move first, and refuse a name that already 

460 # holds a store. `_safe_dataset_name` never picks one, so this is a guard: 

461 # if it ever did, renaming the frames without the annotations would strand 

462 # them under the old name. 

463 import scanpath_studio.annotations as _annotations 

464 

465 if not _annotations.rename_dataset(st.session_state, old, name): 

466 return None 

467 st.session_state["_datasets"] = { 

468 (name if key == old else key): value for key, value in store.items() 

469 } 

470 if st.session_state.get("data_source_choice") == old: 

471 st.session_state["_pending_source_choice"] = name 

472 if st.session_state.get("_prev_source") == old: 

473 st.session_state["_prev_source"] = name 

474 if st.session_state.get(COMPARE_SOURCE_STATE_KEY) == old: 

475 st.session_state[COMPARE_SOURCE_STATE_KEY] = name 

476 # UX-174 r2 — and its description. 

477 descriptions = dict(st.session_state.get(DATASET_DESCRIPTIONS_KEY) or {}) 

478 if old in descriptions: 

479 descriptions[name] = descriptions.pop(old) 

480 st.session_state[DATASET_DESCRIPTIONS_KEY] = descriptions 

481 # DATA-47 — and its metadata tables, which are keyed by the name too (the 

482 # annotations moved above). 

483 from scanpath_studio import metadata as _metadata 

484 

485 _metadata.rename_dataset(st.session_state, old, name) 

486 rename_cached_dataset(st.session_state, old, name) 

487 return name 

488 

489 

490def _render_leave_prompt(host) -> None: 

491 """Ask before abandoning a half-built dataset (BUG-31). 

492 

493 Raised by ``app.main`` when the user navigates away while the wizard is open: 

494 rather than switching the app onto the unfinished upload — which reported 

495 *"this dataset isn't set up yet"* over a session full of finished datasets — 

496 the view is held here and the question is asked. 

497 

498 It says the files will be lost because they will. Streamlit drops a widget's 

499 key at the end of any run in which it did not render, and ``st.file_uploader`` 

500 is the one widget ``persist_state="session"`` cannot cover (ENG-36), so 

501 "park it and come back" is not available to promise. The mapping answers 

502 would survive; the uploads would not, which is the half that matters. 

503 

504 **UX-79** made it a modal. ``host`` is kept in the signature — the caller 

505 reserves a place in the wizard's sticky bar and nothing else needs to know 

506 that the question now opens over the page rather than inside it. 

507 """ 

508 destination = st.session_state.get(WIZARD_LEAVE_KEY) 

509 if not destination: 

510 return 

511 _leave_prompt_dialog(destination) 

512 

513 

514@st.dialog("Leave setup?") 

515@guarded() 

516def _leave_prompt_dialog(destination: str) -> None: 

517 """The modal body — UX-79. Opened by :func:`_render_leave_prompt`. 

518 

519 A dialog rather than a bar inside the wizard: the question is raised by a 

520 click on the **nav**, at the top of the window, while the answer used to 

521 appear inside a screen the user is already scrolled down in. It also has to 

522 interrupt — the run that asks it is the run that would otherwise have 

523 navigated away. 

524 

525 The keys are unchanged (`WIZARD_LEAVE_KEY` arms it; the two callbacks clear 

526 it), so `app.main`'s hold-the-view logic is untouched: a dialog is opened by 

527 *calling* it, so all that moved is where the flag is read. 

528 

529 **BUG-36:** handled by the button's *return value*, not `on_click` — an 

530 `st.dialog` body is a fragment, so an `on_click` callback here reran only 

531 the dialog: it wrote the session state fine, but `main()` never 

532 re-executed, so the modal sat there looking inert. `st.rerun(scope="app")` 

533 both closes the modal and renders the page underneath (see `tour.py`'s 

534 `_tutorial_library_dialog`, which hit the same trap first). 

535 """ 

536 st.warning( 

537 f"**Leave setup and go to {view_label(destination)}?** This dataset isn't " 

538 "added yet " 

539 "— the files you uploaded won't be kept.", 

540 icon=ICONS["warning"], 

541 ) 

542 stay_col, leave_col = st.columns(2) 

543 if stay_col.button( 

544 f"{ICONS['undo']} Keep setting up", 

545 key="wizard_leave_stay", 

546 width="stretch", 

547 type="primary", 

548 ): 

549 app.stay_in_wizard() 

550 st.rerun(scope="app") 

551 if leave_col.button( 

552 f"{ICONS['delete']} Discard and leave", 

553 key="wizard_leave_discard", 

554 width="stretch", 

555 ): 

556 app.discard_and_leave_wizard() 

557 st.rerun(scope="app") 

558 

559 

560def _enter_add_data_wizard() -> None: 

561 """Open the upload wizard (the "➕ Add data" button's ``on_click`` callback). 

562 

563 Tracks the wizard in a *plain* ``_show_upload_wizard`` flag rather than 

564 stuffing ``UPLOAD_CHOICE`` into the ``data_source_choice`` radio key. The 

565 radio isn't rendered while the wizard is open, and Streamlit garbage-collects 

566 a not-rendered widget key after a couple of reruns — which used to silently 

567 drop ``data_source_choice`` mid-wizard (more so for a composite trial id, 

568 which needs more interactions/reruns) and bounce the user back to the main 

569 app. The flag is never GC'd, and the radio's value stays a real source.""" 

570 st.session_state["_prev_source"] = st.session_state.get( 

571 "data_source_choice", DEMO_CHOICE 

572 ) 

573 st.session_state["_show_upload_wizard"] = True 

574 st.session_state["setup_complete"] = False 

575 # DATA-47: a new dataset starts with no metadata tables — not the ones of 

576 # the dataset the wizard was opened over, and not a previous attempt's — 

577 # and no annotations (DATA-48). 

578 import scanpath_studio.annotations as _annotations 

579 from scanpath_studio import metadata as _metadata 

580 

581 _metadata.begin_pending_dataset(st.session_state) 

582 _annotations.begin_pending_dataset(st.session_state) 

583 # DATA-26: the wizard is the 🗂️ Data page's add-a-dataset mode, so take the 

584 # user there. Written as a *request* (`menu.render_nav` reconciles it on the 

585 # next run) rather than a `switch_to_view` — this is an `on_click` callback, 

586 # where Streamlit forbids `st.switch_page`. 

587 st.session_state["main_nav"] = _VIEW_DATA 

588 _reset_wizard_widgets() 

589 

590 

591def _map_section( 

592 raw, specs, proposed, prefix, host, keys, *, per_row: int = 1, stacked: bool = True 

593) -> dict: 

594 """Render a subset of a table's mapping fields. Returns that partial mapping. 

595 

596 ``per_row`` (UX-53 r7) is how many fields share a line — four fixation 

597 fields fit where one used to, which is what gets the whole mapping onto one 

598 screen. ``stacked`` keeps every wizard field's title *above* its control, 

599 including a lone field dropped into a row the caller laid out (r17); the 

600 🗂️ Data page keeps `label | field` by not going through here. 

601 

602 No ``on_change`` any more: the accordion's open state is owned by the keyed 

603 expander and moved only by explicit navigation (``wizard_shell``), so a 

604 mapping widget no longer has to defend the step it lives in against being 

605 collapsed by its own edit (DATA-19's mechanism, deleted by DATA-22).""" 

606 if raw is None or getattr(raw, "empty", True): 

607 return {} 

608 return column_mapping_ui( 

609 raw, 

610 table_label="", 

611 state_key_prefix=prefix, 

612 field_specs=specs, 

613 proposed=proposed, 

614 container=host, 

615 use_expander=False, 

616 only_keys=keys, 

617 header=False, 

618 columns_per_row=per_row, 

619 stack_labels=stacked, 

620 dataset=WIZARD_MAPPING_DATASET, 

621 ) 

622 

623 

624def _default_trial_columns(proposed: dict, present_cols) -> list: 

625 """Default trial-id mapping for the wizard, restricted to ``present_cols`` (the 

626 columns common to every table). 

627 

628 A *trial* is one reading of one passage, so the default composes the 

629 participant with the finest passage identifier present (paragraph preferred 

630 over a coarser text id), plus a repeated-reading column when the data has one 

631 — otherwise the two readings of the same paragraph would collapse into one 

632 trial (OneStop's own ``unique_trial_id`` is participant + paragraph + 

633 repeated reading). When that composite can't be built, prefer a single 

634 precomputed unique trial id over a redundant composite (e.g. don't pair 

635 ``unique_trial_id`` with the paragraph id it already encodes).""" 

636 cols_frame = pd.DataFrame(columns=list(present_cols)) 

637 # Use the canonical candidate lists so non-standard names (reader_id, 

638 # recording_session_label, …) are matched here too, not just by the schema 

639 # auto-detect. 

640 participant = pick_column(cols_frame, PARTICIPANT_CANDIDATES) 

641 paragraph = pick_column(cols_frame, ["unique_paragraph_id", "paragraph_id"]) 

642 text = pick_column(cols_frame, ["unique_text_id", "text_id"]) 

643 # Finest passage grain first: a paragraph is one trial; a text/article may 

644 # span several paragraphs. 

645 passage = paragraph or text 

646 repeated = pick_column(cols_frame, ["repeated_reading_trial", "reread"]) 

647 trial = proposed.get("trial") 

648 trial_present = bool(trial) and trial in set(present_cols) 

649 # A precomputed unique trial id that isn't just the passage we'd otherwise 

650 # compose wins outright (no benefit composing). 

651 if trial_present and trial not in (paragraph, text): 

652 return [trial] 

653 if participant and passage: 

654 return [c for c in (participant, passage, repeated) if c] 

655 if paragraph and text: 

656 return [paragraph, text] 

657 if trial_present: 

658 return [trial] 

659 # No usable composite/trial — fall back to whatever passage component we 

660 # have (never a lone participant, which isn't a trial), else leave empty so 

661 # the user picks. 

662 return list(dict.fromkeys(c for c in (passage,) if c)) 

663 

664 

665# ----------------------------------------------------------------------------- 

666# Cached per-rerun work (DATA-22 §5) 

667# 

668# Every one of these used to recompute on *every keystroke* with no spinner — a 

669# `nunique` over the whole uploaded frame, a full column scan, a re-proposal of 

670# the schema — while the user typed in a text box. They are now keyed on 

671# `data.frame_fingerprint` plus the mapping, and each carries a labelled spinner 

672# so the work is visible rather than felt. Frames pass un-hashed (underscore 

673# args) per the house convention; the fingerprint is the real key. 

674# ----------------------------------------------------------------------------- 

675 

676 

677def _mapping_key(mapping) -> tuple: 

678 """A hashable cache key for a column mapping (str | list | None).""" 

679 if mapping is None: 

680 return () 

681 if isinstance(mapping, (list, tuple)): 

682 return tuple(mapping) 

683 return (mapping,) 

684 

685 

686@st.cache_data(show_spinner="Counting trials…") 

687def _trial_id_values_cached( 

688 _raw, _mapping, fingerprint: tuple, key: tuple 

689) -> frozenset: 

690 return frozenset(trial_id_series(_raw, _mapping).unique()) 

691 

692 

693def _trial_id_values(raw, schema) -> set | None: 

694 """Set of distinct trial-id strings for a raw frame + its trial mapping 

695 (composite mappings are joined, mirroring ``data.trial_id_series``). ``None`` 

696 when the trial isn't mapped or its columns are absent.""" 

697 if raw is None or getattr(raw, "empty", True) or not schema.get("trial"): 

698 return None 

699 cols = trial_mapping_columns(schema["trial"]) 

700 if not cols or not all(c in raw.columns for c in cols): 

701 return None 

702 return set( 

703 _trial_id_values_cached( 

704 raw, 

705 schema["trial"], 

706 frame_fingerprint(raw), 

707 _mapping_key(schema["trial"]), 

708 ) 

709 ) 

710 

711 

712@st.cache_data(show_spinner="Detecting columns…") 

713def _c_propose_word_schema(_raw, fingerprint: tuple) -> dict: 

714 return propose_word_schema(_raw) 

715 

716 

717@st.cache_data(show_spinner="Detecting columns…") 

718def _c_propose_fix_schema(_raw, fingerprint: tuple) -> dict: 

719 return propose_fix_schema(_raw) 

720 

721 

722@st.cache_data(show_spinner="Detecting columns…") 

723def _c_propose_raw_gaze_schema(_raw, fingerprint: tuple) -> dict: 

724 return propose_raw_gaze_schema(_raw) 

725 

726 

727@st.cache_data(show_spinner="Scanning fields…") 

728def _c_categorize_columns(_raw, _schema, _registry, fingerprint: tuple, key: tuple): 

729 # Only ``fingerprint`` + ``key`` form the cache key; the frame, the mapping 

730 # dict and the module-level registry table all ride un-hashed. 

731 return categorize_columns(_raw, _schema, _registry) 

732 

733 

734@st.cache_data(show_spinner="Aggregating character boxes…") 

735def _c_aggregate_char_boxes(_raw, _schema, fingerprint: tuple, key: tuple): 

736 return aggregate_char_boxes(_raw, _schema) 

737 

738 

739def _readers_do_not_line_up(words: pd.DataFrame, fixations: pd.DataFrame) -> bool: 

740 """Whether the normalized tables share trial ids but no (participant, trial) 

741 pair — the reader half of the join is what failed (BUG-59).""" 

742 if words.empty or fixations.empty: 

743 return False 

744 

745 def check() -> bool: 

746 if not set(words["trial_id"]) & set(fixations["trial_id"]): 

747 return False # the trial-id warning already says so 

748 return not trial_keys(words) & trial_keys(fixations) 

749 

750 key = (frame_fingerprint(words), frame_fingerprint(fixations)) 

751 return frame_cache("wizard_readers_line_up", key, check) 

752 

753 

754@st.cache_data(show_spinner=False) 

755def _c_normalization_issues( 

756 _raw, _schema, fingerprint: tuple, key: tuple, table: str 

757) -> list: 

758 return normalization_issues( 

759 _raw, _schema, table=table, fixations=table == "Fixations" 

760 ) 

761 

762 

763def _schema_key(schema: dict | None) -> tuple: 

764 """A hashable, order-stable projection of a mapping dict for cache keys.""" 

765 if not schema: 

766 return () 

767 return tuple( 

768 sorted( 

769 (k, tuple(v) if isinstance(v, (list, tuple)) else v) 

770 for k, v in schema.items() 

771 ) 

772 ) 

773 

774 

775#: The mapping fields the screen estimate reads — its cache keys on these alone, 

776#: so an unrelated pick (the trial id, a kept field) does not re-estimate. 

777_WORD_GEOMETRY_FIELDS = ("left", "right", "top", "bottom", "x", "y", "width", "height") 

778_FIX_GEOMETRY_FIELDS = ("x", "y") 

779 

780 

781def _geometry_key(schema: dict | None, fields: tuple) -> tuple: 

782 return _schema_key({k: (schema or {}).get(k) for k in fields} if schema else None) 

783 

784 

785@st.cache_data(show_spinner=False) 

786def _c_estimate_canvas( 

787 _words, _word_schema, _fix, _fix_schema, fingerprints: tuple, keys: tuple 

788) -> tuple[int, int]: 

789 """DATA-46 — the screen estimate from the mapped geometry of a raw upload. 

790 

791 Cached like the other `_c_*` helpers: the wizard reruns on every keystroke, 

792 and a text-typed coordinate column makes the conversion slow. Returns the 

793 size alone — returning the projected frames would copy them on every hit. 

794 """ 

795 return compute_canvas_size( 

796 *canvas_geometry_frames(_words, _word_schema, _fix, _fix_schema) 

797 ) 

798 

799 

800def _render_identity_field( 

801 field_key: str, 

802 label: str, 

803 help_text: str, 

804 cells, 

805 raw_words, 

806 raw_fix, 

807 word_schema, 

808 fix_schema, 

809 has_words, 

810 has_fix, 

811 default_cols: list, 

812 required: bool = False, 

813) -> None: 

814 """One identifier (trial / participant / text), **one picker per table**. 

815 

816 UX-53 r13 removed the *Different … per table* toggle. It was a mode switch 

817 guarding a case that costs nothing to show: the tables are listed anyway, so 

818 naming the column in each is one pick either way, and the toggle made the 

819 common case ("same column, obviously") look like a decision while hiding the 

820 uncommon one behind a control nobody finds. Each present table now gets its 

821 own multiselect, seeded from the same proposal, so identical columns are 

822 still identical — just visible. 

823 

824 Several columns compose an id (joined with ``_``, like the trial id). Writes 

825 ``schema[field_key]`` (str / list / None) into ``word_schema`` / 

826 ``fix_schema`` in place; the per-table ``col_map_<tbl>_<field>`` keys are the 

827 stored values, which is what save/restore and deep links already carry. 

828 

829 ``cells`` is one container per present table, in (fixations, words) order — 

830 the caller lays the row out, because a theme's fields share one row. 

831 """ 

832 

833 def _mapping(chosen): 

834 if not chosen: 

835 return None 

836 return chosen[0] if len(chosen) == 1 else list(chosen) 

837 

838 def _seed(key: str, options: list, fallback: list) -> None: 

839 """Seed a multiselect's session value, dropping columns absent from 

840 ``options`` (a new upload changes the column universe), like 

841 ``column_mapping_ui`` — so the stale-reset never fights a default arg.""" 

842 stored = st.session_state.get(key) 

843 if stored is None: 

844 st.session_state[key] = [c for c in fallback if c in options] 

845 return 

846 valid = [c for c in stored if c in options] 

847 if len(valid) != len(stored): 

848 st.session_state[key] = valid or [c for c in fallback if c in options] 

849 

850 tables = [] 

851 if has_fix: 

852 tables.append(("fix", "Fixations", raw_fix, fix_schema)) 

853 if has_words: 

854 tables.append(("words", WORDS_TABLE_LABEL, raw_words, word_schema)) 

855 

856 tinted: dict[str, list[str]] = {} 

857 for cell, (slug, table_label, raw, schema) in zip(cells, tables): 

858 key = f"col_map_{slug}_{field_key}" 

859 # UX-108 — the file's real header, not `raw.columns`: PERF-6 parses 

860 # only the columns an auto-detect + the optional-field registry + 

861 # whatever a `col_map_*` key already names decided to keep, and Trial 

862 # ID / Participant ID / Text ID are exactly the fields a composite 

863 # mapping is built from — the ones a narrow parse is most likely to 

864 # have left out. Same fallback as `controls.column_mapping_ui`: empty 

865 # outside a real upload (this function also renders for a *stored* 

866 # dataset's already-normalized frames, which have no header to ask). 

867 full_header = st.session_state.get(f"col_map_{slug}_header") 

868 options = list(full_header) if full_header else list(raw.columns) 

869 _seed(key, options, default_cols) 

870 # Title on the cell's first line, control on its second (UX-53 r14), so 

871 # the row reads as a line of names over a line of pickers. The title 

872 # carries no table name (r15): the rows are now grouped *by* table and 

873 # each is labelled once at its head, so repeating it on all three fields 

874 # would say the same thing three times. 

875 # UX-176: the title shares its line with the ✨ flag, as a select's 

876 # does, so an auto-detected id can be seen and confirmed. 

877 label_col, flag_col = cell.container().columns( 

878 _GRID_LABEL_W, gap=None, vertical_alignment="center" 

879 ) 

880 inline_field_label(label_col, label, f"{help_text} ({table_label} table)") 

881 # UX-91: a keyed wrapper so an empty *required* picker can be tinted the 

882 # red every other required field turns after a failed add. These 

883 # multiselects are the wizard's own — they never went through 

884 # `column_mapping_ui`, which is why Trial ID stayed grey while the 

885 # selects beside it went red. 

886 cell_key = f"{key}_cell" 

887 chosen = cell.container(key=cell_key).multiselect( 

888 f"{label} — {table_label}", 

889 options=options, 

890 key=key, 

891 help=help_text, 

892 label_visibility="collapsed", 

893 # #374 F13: an id is one column or a few, never every column. 

894 select_all=False, 

895 on_change=_mark_field_touched, 

896 args=(key,), 

897 ) 

898 schema[field_key] = _mapping(chosen) 

899 state = multi_field_flag( 

900 flag_col, 

901 state_key=key, 

902 cell_key=cell_key, 

903 chosen=list(chosen), 

904 default=[c for c in default_cols if c in options], 

905 required=required, 

906 preview=value_preview_tip(raw, field_key, list(chosen)), 

907 ) 

908 if state: 

909 tinted.setdefault(state, []).append(cell_key) 

910 mark_cells(tinted) 

911 

912 

913#: UX-113 — session key holding the *committed* filename-derive settings 

914#: (mode/delimiter/pattern/lower). Kept separate from the live widget keys 

915#: below so editing the delimiter or regex doesn't silently change what is 

916#: actually derived — only clicking Apply does. 

917#: UX-129: a list of one dict per derivation line (➕ Another adds lines; 

918#: Apply commits all of them at once) — a plain dict is still accepted when 

919#: read back (a setup saved before UX-129 wrote one), see the normalization 

920#: where this is consumed. 

921_FILENAME_DERIVE_APPLIED_KEY = "wizard_filename_applied" 

922 

923#: UX-129 — how many derivation lines are drawn (➕ Another increments it). 

924#: Session-only: a restored setup replays the *committed* `applied` list 

925#: regardless of this count, so a saved multi-line derivation still takes 

926#: effect even though re-opening the control to edit it starts back at one 

927#: line (see `_filename_derive_section`). 

928_FILENAME_DERIVE_LINE_COUNT_KEY = "wizard_filename_line_count" 

929 

930#: A worked example rather than a blank box — shown as the regex field's own 

931#: starting value (edit or replace it) and repeated in its help text so the 

932#: two stay in sync. 

933_FILENAME_REGEX_EXAMPLE = r"(?P<session>\d+_\w+_ET\d)_.*_(?P<stimulus>.+)_scanpath" 

934 

935#: UX-129 — the three tables a derivation line can target, display label to 

936#: the `col_map_*_header` prefix its derived columns publish into 

937#: (`_wizard_filename_derive`'s "make it mappable below" step). 

938_DERIVE_TABLE_PREFIX = { 

939 "Fixations": "col_map_fix", 

940 "Words / IA": "col_map_words", 

941 "Raw gaze": "col_map_raw_gaze", 

942} 

943 

944#: #374 F12 — the word table's one name on the add and edit screens. Issue 

945#: #375 may revisit it; every label and message reads it from here. 

946WORDS_TABLE_LABEL = "Words (interest areas)" 

947 

948#: How the derive picker shows each table (its values stay the keys above, 

949#: which saved setups carry). 

950_DERIVE_TABLE_DISPLAY = {"Words / IA": WORDS_TABLE_LABEL} 

951 

952#: #374 F12 — the one-line caption above each upload row: what goes there, in 

953#: EyeLink's terms. 

954ROW_CAPTIONS = { 

955 "fixations": "One row per fixation — e.g. EyeLink's Fixation Report.", 

956 "words": "One row per word with its box — e.g. EyeLink's Interest Area Report.", 

957} 

958 

959 

960def _row_note(block, text: str): 

961 """A caption line across the top of an upload row's mapping side (#374 

962 F12), returned so later notes for the row (a mixed ZIP's) land under it. 

963 

964 Drawn right of the row's name column: that column is an overlay spanning 

965 the whole block (`styles.py`), so a full-width line would sit under it.""" 

966 note = block.columns(_META_ROW_W, gap="small")[1] 

967 note.caption(text) 

968 return note 

969 

970 

971#: UX-129 — one shared column-width tuple for every derivation line (Table · 

972#: Column · How · pattern · Lowercase · ➕ Another · Apply), so the five 

973#: shared selectors are pixel-identical whether it's line 0 (which also 

974#: carries the two buttons) or a later line (which leaves those two slots 

975#: empty) — same `st.columns` call, same weights, every row. 

976_DERIVE_LINE_W = (0.14, 0.16, 0.15, 0.24, 0.09, 0.11, 0.11) 

977 

978 

979def _next_available_names(existing: set, base: str, count: int) -> list: 

980 """The next ``count`` names ``<base>_<n>`` not already in ``existing``, 

981 marking each as claimed as it is picked (UX-129). 

982 

983 Splitting the same column a second time then continues past whatever the 

984 first split already claimed (``source_file_7``, not another 

985 ``source_file_1``) instead of colliding with — and silently 

986 overwriting — it. ``existing`` is mutated in place so a caller reusing it 

987 across several lines against the same table keeps accumulating names. 

988 """ 

989 names = [] 

990 n = 1 

991 while len(names) < count: 

992 candidate = f"{base}_{n}" 

993 if candidate not in existing: 

994 names.append(candidate) 

995 existing.add(candidate) 

996 n += 1 

997 return names 

998 

999 

1000#: The last derivation each Apply line made, per table: what it was made from 

1001#: and the frame it gave. A derivation is a pure function of its input frame and 

1002#: its settings, and the add screen reruns on every click — so without this a 

1003#: multi-million-row raw-gaze table was split (or regex-matched) again on each 

1004#: one. One slot per (table, line), replaced when its input or settings change. 

1005_FILENAME_DERIVE_MEMO_KEY = "_wizard_filename_derive_memo" 

1006 

1007 

1008def _memo_derive(slot: tuple, key: tuple, compute): 

1009 """``compute()``'s result for ``key``, reused while ``key`` is unchanged. 

1010 

1011 ``key`` holds the input frame's fingerprint (`data.frame_fingerprint`, an 

1012 assigned ID for an uploaded table, so looking it up hashes nothing) and the 

1013 settings, so new data or a new pattern always derives afresh. 

1014 """ 

1015 memo = st.session_state.setdefault(_FILENAME_DERIVE_MEMO_KEY, {}) 

1016 hit = memo.get(slot) 

1017 if hit is not None and hit[0] == key: 

1018 return hit[1] 

1019 result = compute() 

1020 memo[slot] = (key, result) 

1021 return result 

1022 

1023 

1024def _wizard_filename_derive(body, raw_words, raw_fix, raw_gaze): 

1025 """Optional step: derive columns from an uploaded column's own text — the 

1026 captured ``source_file`` by default, or any other column — so identity 

1027 that lives only inside a string (a filename, or a structured value like 

1028 ``question_04111_target``) can be mapped. 

1029 

1030 UX-129 — each line picks its **Table** first, then a **Column** within 

1031 it: Fixations/Words/Raw gaze each carry their own `source_file` (one row 

1032 per upload, not one shared value), so a line derives from exactly the one 

1033 table named, never all three at once the way the original single-source 

1034 version did. Two modes per line: split into positional 

1035 ``<column>_1``/``<column>_2``/… columns on a delimiter (collision-safe — 

1036 see :func:`_next_available_names`), or extract *named groups* with a 

1037 regex (robust to variable-length parts, e.g. a stimulus name whose length 

1038 varies). ➕ **Another** (beside Apply, on the first line) adds another 

1039 mapping line — the same Table/Column/How/pattern/Lowercase controls, no 

1040 Apply of its own — for when identity has to be pulled from more than one 

1041 place (a session id from Fixations' filename *and* a question id from 

1042 Words' own `page` column, say). Nothing is derived, and nothing becomes 

1043 mappable below, until **Apply**, which commits every line at once — the 

1044 settings here are a draft until then, so retyping a regex mid-thought 

1045 never silently changes the committed columns. Returns the (possibly 

1046 augmented) frames so the identifier pickers below see the new columns; a 

1047 no-op while the toggle is off. 

1048 """ 

1049 toggle_col, controls_col = body.columns( 

1050 [0.26, 0.74], gap="small", vertical_alignment="center" 

1051 ) 

1052 enabled = toggle_col.toggle( 

1053 "Derive columns from text", 

1054 key="wizard_filename_split", 

1055 help=( 

1056 "When identity lives only inside a text value — no column carries " 

1057 "it on its own — derive real columns from it (e.g. session, " 

1058 "stimulus) that you can then map as an id below: split it on a " 

1059 "delimiter into positional columns, or pull out named regex " 

1060 "groups for parts of variable length (e.g. a stimulus name). " 

1061 "Pick the table first, then any of its own columns — defaults to " 

1062 f"the uploaded filename (captured as `source_file`). {ICONS['add']} Add line " 

1063 "adds a second line when one column's text isn't enough." 

1064 ), 

1065 ) 

1066 if not enabled: 

1067 return raw_words, raw_fix, raw_gaze 

1068 

1069 # UX-129: every table, always — including empty ones — so a line whose 

1070 # saved config named a table that has since lost its upload still has a 

1071 # defined (if inert) frame to look up below, no special-casing needed. 

1072 frames = {"Fixations": raw_fix, "Words / IA": raw_words, "Raw gaze": raw_gaze} 

1073 table_options = [label for label, fr in frames.items() if not fr.empty] or [ 

1074 "Fixations" 

1075 ] 

1076 

1077 # UX-129: one to many lines. Line 0 keeps the original, unsuffixed widget 

1078 # keys (`wizard_filename_column`, not `_0`) — a saved setup written 

1079 # before this existed seeds exactly those, and the single-line shape is 

1080 # still the common case. `another_col`/`apply_col` are reserved on line 

1081 # 0's own row and filled in *after* every line has been drawn (the 

1082 # reserve-then-fill trick this module uses throughout), since Apply's 

1083 # disabled state depends on every line's draft, not just the first. 

1084 line_count = max(1, int(st.session_state.get(_FILENAME_DERIVE_LINE_COUNT_KEY, 1))) 

1085 another_col = apply_col = None 

1086 drafts: list[dict] = [] 

1087 for idx in range(line_count): 

1088 suffix = "" if idx == 0 else f"_{idx}" 

1089 table_key = f"wizard_filename_table{suffix}" 

1090 column_key = f"wizard_filename_column{suffix}" 

1091 mode_key = f"wizard_filename_mode{suffix}" 

1092 delim_key = f"wizard_filename_delim{suffix}" 

1093 regex_key = f"wizard_filename_regex{suffix}" 

1094 lower_key = f"wizard_filename_regex_lower{suffix}" 

1095 

1096 if st.session_state.get(table_key) not in table_options: 

1097 st.session_state[table_key] = table_options[0] 

1098 

1099 ( 

1100 table_col, 

1101 col_col, 

1102 mode_col, 

1103 input_col, 

1104 lower_col, 

1105 another_col, 

1106 apply_col, 

1107 ) = controls_col.columns( 

1108 _DERIVE_LINE_W, gap="small", vertical_alignment="bottom" 

1109 ) 

1110 selected_table = table_col.selectbox( 

1111 "Table", 

1112 table_options, 

1113 key=table_key, 

1114 label_visibility="collapsed", 

1115 help="Which uploaded table to derive from — its own columns are " 

1116 "offered next, since two tables' `source_file` rarely mean the " 

1117 "same thing.", 

1118 # #374 F12: the value stays "Words / IA" — saved setups name it. 

1119 format_func=lambda t: _DERIVE_TABLE_DISPLAY.get(t, t), 

1120 ) 

1121 table_frame = frames[selected_table] 

1122 table_columns = ( 

1123 list(table_frame.columns) if not table_frame.empty else [SOURCE_FILE_COLUMN] 

1124 ) 

1125 default_column = ( 

1126 SOURCE_FILE_COLUMN 

1127 if SOURCE_FILE_COLUMN in table_columns 

1128 else table_columns[0] 

1129 ) 

1130 if st.session_state.get(column_key) not in table_columns: 

1131 st.session_state[column_key] = default_column 

1132 

1133 source_column = col_col.selectbox( 

1134 "Column", 

1135 table_columns, 

1136 key=column_key, 

1137 label_visibility="collapsed", 

1138 help="Which of that table's own columns to derive from — " 

1139 "defaults to the filename (`source_file`); pick any other " 

1140 "column that encodes identity as text.", 

1141 ) 

1142 mode = mode_col.selectbox( 

1143 "How", 

1144 ["Split on a delimiter", "Regex named groups"], 

1145 key=mode_key, 

1146 label_visibility="collapsed", 

1147 help="How to pull columns out of the chosen column's text.", 

1148 ) 

1149 pattern = "" 

1150 lower = False 

1151 delimiter = "_" 

1152 if mode == "Split on a delimiter": 

1153 delimiter = ( 

1154 input_col.text_input( 

1155 "Delimiter", 

1156 value="_", 

1157 key=delim_key, 

1158 max_chars=8, 

1159 label_visibility="collapsed", 

1160 help="Character(s) to split the text on — e.g. " 

1161 "`reader0_b0_scanpath` → reader0 / b0 / scanpath.", 

1162 ) 

1163 or "_" 

1164 ) 

1165 else: 

1166 pattern = input_col.text_input( 

1167 "Regex (named groups)", 

1168 value=_FILENAME_REGEX_EXAMPLE, 

1169 key=regex_key, 

1170 label_visibility="collapsed", 

1171 help=f"e.g. `{_FILENAME_REGEX_EXAMPLE}` — each named group becomes " 

1172 "a column. Edit this to match your own filenames.", 

1173 ) 

1174 lower = lower_col.toggle( 

1175 "Lowercase", 

1176 key=lower_key, 

1177 help="Fold case so a value matches across tables (e.g. CamelCase " 

1178 "scanpath names vs lowercase AOI names).", 

1179 ) 

1180 if pattern: 

1181 try: 

1182 re.compile(pattern) 

1183 except re.error as exc: 

1184 controls_col.error(f"Invalid pattern: {exc}") 

1185 pattern = "" 

1186 

1187 drafts.append( 

1188 { 

1189 "table": selected_table, 

1190 "mode": mode, 

1191 "column": source_column, 

1192 "delimiter": delimiter if mode == "Split on a delimiter" else None, 

1193 "pattern": pattern if mode != "Split on a delimiter" else None, 

1194 "lower": lower, 

1195 } 

1196 ) 

1197 

1198 if another_col.button( 

1199 f"{ICONS['add']} Add line", 

1200 key="wizard_filename_another", 

1201 width="stretch", 

1202 help="Add another mapping line — for when identity has to be pulled " 

1203 "from more than one column (or table), or the same column two " 

1204 "different ways.", 

1205 ): 

1206 # UX-129: `line_count` was already read (and the lines above already 

1207 # drawn with it) before this button was evaluated, so raising it here 

1208 # alone would only take effect on the *next* interaction — a click 

1209 # that visibly did nothing. `st.rerun()` restarts the script now, so 

1210 # the new line appears immediately, same run. 

1211 st.session_state[_FILENAME_DERIVE_LINE_COUNT_KEY] = line_count + 1 

1212 st.rerun() 

1213 apply_clicked = apply_col.button( 

1214 "Apply", 

1215 key="wizard_filename_apply", 

1216 width="stretch", 

1217 disabled=any( 

1218 d["mode"] == "Regex named groups" and not d["pattern"] for d in drafts 

1219 ), 

1220 help="Derive the columns above and make them available to map below.", 

1221 ) 

1222 if apply_clicked: 

1223 st.session_state[_FILENAME_DERIVE_APPLIED_KEY] = drafts 

1224 

1225 applied = st.session_state.get(_FILENAME_DERIVE_APPLIED_KEY) 

1226 if not applied: 

1227 return raw_words, raw_fix, raw_gaze 

1228 # UX-129: a setup saved before multiple lines (or per-line tables) 

1229 # existed committed one dict with no "table" key; a current Apply always 

1230 # writes a list where every entry has one. Normalize so both process 

1231 # through the same loop below. 

1232 applied_configs = applied if isinstance(applied, list) else [applied] 

1233 

1234 # UX-129: seeded from the ORIGINAL upload, before any config in this pass 

1235 # has run, then grown as each config below claims names for its table — 

1236 # what makes a second split of the same column continue numbering rather 

1237 # than reusing (and silently overwriting) the first split's own columns. 

1238 existing_by_table = {label: set(fr.columns) for label, fr in frames.items()} 

1239 

1240 for line, config in enumerate(applied_configs): 

1241 # Older sessions' applied state predates the column picker (UX-113) — 

1242 # `source_file` was the only option then, so that's the correct 

1243 # fallback. 

1244 applied_column = config.get("column", SOURCE_FILE_COLUMN) 

1245 target_table = config.get("table") 

1246 if target_table not in frames: 

1247 # UX-129: a config saved before per-line tables existed didn't 

1248 # record one — the old broadcast-to-all-three behavior applied 

1249 # the same split/regex to every table that had the column; a 

1250 # derivation now targets exactly one, so the first match wins. 

1251 target_table = next( 

1252 (label for label, fr in frames.items() if applied_column in fr.columns), 

1253 None, 

1254 ) 

1255 if target_table is None: 

1256 continue 

1257 target = frames[target_table] 

1258 if target.empty or applied_column not in target.columns: 

1259 continue 

1260 existing = existing_by_table.setdefault(target_table, set()) 

1261 parent = target 

1262 

1263 slot = (target_table, line) 

1264 if config["mode"] == "Split on a delimiter": 

1265 

1266 def _split( 

1267 parent=parent, 

1268 existing=frozenset(existing), 

1269 delimiter=config["delimiter"], 

1270 column=applied_column, 

1271 ): 

1272 split = split_source_file(parent, delimiter=delimiter, column=column) 

1273 # UX-129: `split_source_file` always names positionally 

1274 # (`file_part_N`) — rename to `<source column>_<n>` here, using 

1275 # the next names not already claimed in this table, rather than 

1276 # letting a second split of the same (or another) column 

1277 # collide with the first's `file_part_1`/`_2`/…. 

1278 temp_cols = [c for c in split.columns if c not in parent.columns] 

1279 new_names = _next_available_names(set(existing), column, len(temp_cols)) 

1280 derived = split.rename(columns=dict(zip(temp_cols, new_names))) 

1281 # BUG-103: named by what made it, so it is not re-hashed each rerun. 

1282 assign_derived( 

1283 derived, 

1284 "split_source_file", 

1285 parent, 

1286 (delimiter, column, tuple(new_names)), 

1287 ) 

1288 return derived, new_names 

1289 

1290 target, new_cols = _memo_derive( 

1291 slot, 

1292 ( 

1293 "split", 

1294 frame_fingerprint(parent), 

1295 config["delimiter"], 

1296 applied_column, 

1297 tuple(sorted(existing)), 

1298 ), 

1299 _split, 

1300 ) 

1301 existing.update(new_cols) 

1302 else: 

1303 applied_pattern = config.get("pattern") 

1304 if not applied_pattern: 

1305 continue 

1306 # A group named after an existing column would be skipped (real 

1307 # data wins) — flag it so the user renames the group instead of 

1308 # silently getting the original column. 

1309 collisions = source_file_regex_collisions( 

1310 target, applied_pattern, column=applied_column 

1311 ) 

1312 if collisions: 

1313 body.warning( 

1314 "These group names already exist as columns and were left " 

1315 f"untouched: {', '.join(sorted(collisions))}. Rename the " 

1316 "group(s) and Apply again to extract them." 

1317 ) 

1318 

1319 def _extract( 

1320 parent=parent, 

1321 pattern=applied_pattern, 

1322 column=applied_column, 

1323 lower=bool(config["lower"]), 

1324 ): 

1325 derived = extract_columns_from_source_file( 

1326 parent, pattern, column=column, lowercase=lower 

1327 ) 

1328 assign_derived( 

1329 derived, 

1330 "extract_columns_from_source_file", 

1331 parent, 

1332 (pattern, column, lower), 

1333 ) 

1334 return derived 

1335 

1336 target = _memo_derive( 

1337 slot, 

1338 ( 

1339 "regex", 

1340 frame_fingerprint(parent), 

1341 applied_pattern, 

1342 applied_column, 

1343 bool(config["lower"]), 

1344 ), 

1345 _extract, 

1346 ) 

1347 new_cols = [ 

1348 c for c in re.compile(applied_pattern).groupindex if c in target.columns 

1349 ] 

1350 existing.update(new_cols) 

1351 

1352 frames[target_table] = target 

1353 if new_cols: 

1354 # UX-113: publish the derived columns into the table's own header 

1355 # stash. `column_mapping_ui` offers `col_map_<slug>_header` over 

1356 # the live frame's own columns (PERF-6's narrow-parse plan), 

1357 # which is the *file's* header stashed at upload time — so a 

1358 # column synthesized in memory afterwards, like these, never 

1359 # showed up in the picker below even though the frame itself 

1360 # already carried it. 

1361 header_key = f"{_DERIVE_TABLE_PREFIX[target_table]}_header" 

1362 header = list(st.session_state.get(header_key) or target.columns) 

1363 missing = [c for c in new_cols if c not in header] 

1364 if missing: 

1365 st.session_state[header_key] = [*header, *missing] 

1366 body.caption( 

1367 f"Derived from **{target_table} · {applied_column}** " 

1368 "(first rows) — map them as ids below:" 

1369 ) 

1370 body.dataframe( 

1371 target[[applied_column, *new_cols]].drop_duplicates().head(), 

1372 width="stretch", 

1373 hide_index=True, 

1374 ) 

1375 

1376 raw_fix = frames["Fixations"] 

1377 raw_words = frames["Words / IA"] 

1378 raw_gaze = frames["Raw gaze"] 

1379 return raw_words, raw_fix, raw_gaze 

1380 

1381 

1382def _wizard_trial_step( 

1383 body, 

1384 raw_words, 

1385 raw_fix, 

1386 prop_w, 

1387 prop_f, 

1388 word_schema, 

1389 fix_schema, 

1390 has_words, 

1391 has_fix, 

1392 *, 

1393 cells=None, 

1394) -> str | None: 

1395 """Trial-identifier wizard step: one picker per table (UX-53 r13), plus the 

1396 per-table trial-count check that flags mismatches. Mutates ``word_schema`` / 

1397 ``fix_schema`` in place. ``cells`` are the row containers the caller built — 

1398 each table's count is a caption inside its own cell (UX-67 r2). 

1399 

1400 Returns the one thing that is not a count — the warning that the two 

1401 tables' trial ids do not line up at all — for the caller to draw once the 

1402 Participant ID is mapped too: an AOI table with no Participant ID is 

1403 stimulus-level and attaches by Text ID when the trial ids differ, and the 

1404 wizard states that join (or refuses it) above Add dataset instead (DATA-49). 

1405 """ 

1406 # Core tables present (raw-gaze keeps its own mapping in its own step). 

1407 core = [f for f, present in ((raw_fix, has_fix), (raw_words, has_words)) if present] 

1408 common_cols = [c for c in core[0].columns if all(c in f.columns for f in core)] 

1409 prop_primary = prop_f if has_fix else prop_w 

1410 default_trial = _default_trial_columns(prop_primary, common_cols) 

1411 _render_identity_field( 

1412 "trial", 

1413 "Trial ID *", 

1414 "The column telling one participant's trials apart (e.g. TRIAL_INDEX) — " 

1415 "or several to build one, joined with `_`. Unique within a participant.", 

1416 cells if cells is not None else [body] * 2, 

1417 raw_words, 

1418 raw_fix, 

1419 word_schema, 

1420 fix_schema, 

1421 has_words, 

1422 has_fix, 

1423 default_trial, 

1424 required=True, 

1425 ) 

1426 

1427 # Per-table trial-id sets. Equal → one clean count. Differing but overlapping 

1428 # is usually benign (one table simply covers extra trials, e.g. words for a 

1429 # paragraph with no fixations) → emphasise the difference without implying a 

1430 # mapping error. Differing AND disjoint means the ids don't line up at all. 

1431 sets = {} 

1432 if has_fix: 

1433 sets["Fixations"] = _trial_id_values(raw_fix, fix_schema) 

1434 if has_words: 

1435 sets["Words"] = _trial_id_values(raw_words, word_schema) 

1436 # UX-67 r2: the count is a caption under the picker it counts, not a banner. 

1437 # One `st.success` per identifier stacked three coloured boxes onto a screen 

1438 # whose whole point is that the mapping fits on it — and put the number far 

1439 # from the menu that produces it. Per *table*, too: each cell counts its own 

1440 # column, so a mismatch is read by comparing two numbers side by side rather 

1441 # than by parsing a sentence about it. 

1442 cell_by_table = dict(zip(sets, cells if cells is not None else [])) 

1443 for table, values in sets.items(): 

1444 cell = cell_by_table.get(table) 

1445 if values is None or cell is None: 

1446 continue 

1447 cell.caption(plural(len(values), "Trial ID")) 

1448 # Only a real problem still gets a box, and it renders where UX-67 put the 

1449 # blockers: directly above **Add dataset**. 

1450 present = {k: v for k, v in sets.items() if v is not None} 

1451 if len(present) > 1: 

1452 values = list(present.values()) 

1453 counts_str = ", ".join(f"{k}: **{len(v):,}**" for k, v in present.items()) 

1454 if not set.intersection(*values): 

1455 return ( 

1456 f"{ICONS['warning']} No trial ids are shared across tables — {counts_str}. Check " 

1457 "that each table's **Trial ID** picks the column that names the " 

1458 "same trials." 

1459 ) 

1460 return None 

1461 

1462 

1463@st.cache_data(show_spinner="Counting…") 

1464def _distinct_id_count_cached(_raw, _mapping, fingerprint: tuple, key: tuple) -> int: 

1465 return int(trial_id_series(_raw, _mapping).nunique()) 

1466 

1467 

1468def _distinct_id_count(raw, mapping) -> int | None: 

1469 """Distinct values of a single-column or composite identifier mapping.""" 

1470 if raw is None or getattr(raw, "empty", True) or not mapping: 

1471 return None 

1472 cols = trial_mapping_columns(mapping) 

1473 if not cols or not all(c in raw.columns for c in cols): 

1474 return None 

1475 return _distinct_id_count_cached( 

1476 raw, mapping, frame_fingerprint(raw), _mapping_key(mapping) 

1477 ) 

1478 

1479 

1480def _wizard_participant_text_step( 

1481 field_key: str, 

1482 label: str, 

1483 noun: str, 

1484 help_text: str, 

1485 body, 

1486 raw_words, 

1487 raw_fix, 

1488 prop_w, 

1489 prop_f, 

1490 word_schema, 

1491 fix_schema, 

1492 has_words, 

1493 has_fix, 

1494 *, 

1495 cells=None, 

1496 extras_host=None, 

1497) -> None: 

1498 """Optional participant- or text-identifier step: one picker per table 

1499 (UX-53 r13), then a distinct-value count captioned under **each** table's 

1500 own picker (UX-67 r2) — every present table gets one, the same per-table 

1501 shape `_wizard_trial_step` uses (BUG-35: previously only Fixations' count 

1502 was ever shown, so the AOI row's pickers had nothing under them). Mutates 

1503 the schemas in place.""" 

1504 core = [f for f, present in ((raw_fix, has_fix), (raw_words, has_words)) if present] 

1505 common_cols = [c for c in core[0].columns if all(c in f.columns for f in core)] 

1506 prop_primary = prop_f if has_fix else prop_w 

1507 default = prop_primary.get(field_key) 

1508 default_cols = [default] if default in common_cols else [] 

1509 _render_identity_field( 

1510 field_key, 

1511 label, 

1512 help_text, 

1513 cells if cells is not None else [body] * 2, 

1514 raw_words, 

1515 raw_fix, 

1516 word_schema, 

1517 fix_schema, 

1518 has_words, 

1519 has_fix, 

1520 default_cols, 

1521 ) 

1522 tables = [] 

1523 if has_fix: 

1524 tables.append(("fix", raw_fix, fix_schema)) 

1525 if has_words: 

1526 tables.append(("words", raw_words, word_schema)) 

1527 cell_by_table = dict(zip((slug for slug, *_ in tables), cells)) if cells else {} 

1528 fallback_cell = extras_host if extras_host is not None else body 

1529 for slug, frame, schema in tables: 

1530 n = _distinct_id_count(frame, schema.get(field_key)) 

1531 if n is not None: 

1532 # UX-67 r2: under the picker that produced it, as small text 

1533 # rather than a banner. 

1534 cell_by_table.get(slug, fallback_cell).caption(plural(n, noun)) 

1535 

1536 

1537def _clean_multiselect_state(key: str, valid) -> None: 

1538 """Drop session values for a multiselect that aren't valid options (e.g. a 

1539 restored config from different data), so Streamlit doesn't raise on render.""" 

1540 stored = st.session_state.get(key) 

1541 if isinstance(stored, (list, tuple)): 

1542 valid_set = set(valid) 

1543 cleaned = [v for v in stored if v in valid_set] 

1544 if len(cleaned) != len(stored): 

1545 st.session_state[key] = cleaned 

1546 

1547 

1548def wide_frame_warning(n_extra_fields: int, n_rows: int) -> str | None: 

1549 """PERF-2 warning for selections large enough to affect rerun latency.""" 

1550 if n_extra_fields < 50 and n_extra_fields * max(n_rows, 0) < 5_000_000: 

1551 return None 

1552 return ( 

1553 f"Keeping {n_extra_fields} additional fields across up to {n_rows:,} rows " 

1554 "can slow the app noticeably. Keep only the measures and metadata you " 

1555 "plan to use." 

1556 ) 

1557 

1558 

1559def _wizard_reader_ids(raw_words, word_schema, raw_fix, fix_schema) -> list: 

1560 """The reader ids the finished dataset will have, read from the raw tables. 

1561 

1562 DATA-20's *About your readers* step runs **before** the frames are 

1563 normalized (normalization needs the keep-columns the next step chooses), so 

1564 the join report can't be built from `metadata.participant_ids`. The mapped 

1565 participant column is only *renamed* by `normalize_*` — its values are 

1566 untouched — so reading it off the raw frame gives the same id set the join 

1567 will see, which is what makes "no row for reader p7" trustworthy here. 

1568 

1569 Cached and **gated**, following `app._cached_participant_ids`: this is a 

1570 `.unique()` over the *raw* (unfiltered, un-normalized) frames, the largest 

1571 version there is, and the wizard reruns on every keystroke across all seven 

1572 steps. Only the join report consumes the list, so nothing is scanned until a 

1573 table is actually being attached. 

1574 """ 

1575 from scanpath_studio import metadata as metadata_mod 

1576 

1577 if st.session_state.get(metadata_mod.RAW_SESSION_KEY) is None: 

1578 return [] 

1579 found: set = set() 

1580 for frame, schema in ((raw_fix, fix_schema), (raw_words, word_schema)): 

1581 column = (schema or {}).get("participant") 

1582 if not column or frame is None or frame.empty or column not in frame.columns: 

1583 continue 

1584 found |= set(_wizard_reader_ids_cached(frame, column, frame_fingerprint(frame))) 

1585 return sorted(found) 

1586 

1587 

1588def _wizard_trial_combos(raw_words, word_schema, raw_fix, fix_schema): 

1589 """The ``(participant_id, trial_id)`` pairs the finished dataset will have. 

1590 

1591 DATA-29's table attaches in the wizard, which runs **before** the frames are 

1592 normalized — so, exactly as :func:`_wizard_reader_ids` does one grain up, 

1593 the keys are read off the raw tables through the mapping the user has just 

1594 made. ``trial_id_series`` is what `normalize_*` itself uses to compose a 

1595 trial id, including the multi-column case, so the pairs here are the pairs 

1596 the join will really see. 

1597 

1598 An unmapped participant column yields an empty reader half rather than no 

1599 rows: the dataset has no reader identity, so a trial-keyed table still joins 

1600 and a reader-keyed one correctly matches nothing. 

1601 

1602 Gated on a table actually being attached, and cached per frame — this 

1603 walks the whole raw frame, and the wizard reruns on every keystroke. 

1604 """ 

1605 from scanpath_studio import metadata as metadata_mod 

1606 

1607 empty = pd.DataFrame(columns=["participant_id", "trial_id"]) 

1608 if st.session_state.get(metadata_mod.TRIAL_RAW_SESSION_KEY) is None: 

1609 return empty 

1610 parts = [] 

1611 for frame, schema in ((raw_fix, fix_schema), (raw_words, word_schema)): 

1612 trial_mapping = (schema or {}).get("trial") 

1613 if not trial_mapping or frame is None or frame.empty: 

1614 continue 

1615 columns = trial_mapping_columns(trial_mapping) 

1616 if any(column not in frame.columns for column in columns): 

1617 continue 

1618 participant = (schema or {}).get("participant") 

1619 if participant and participant not in frame.columns: 

1620 participant = None 

1621 parts.append( 

1622 _wizard_trial_combos_cached( 

1623 frame, 

1624 tuple(columns), 

1625 participant, 

1626 frame_fingerprint(frame), 

1627 ) 

1628 ) 

1629 if not parts: 

1630 return empty 

1631 return pd.concat(parts, ignore_index=True).drop_duplicates() 

1632 

1633 

1634@st.cache_data(show_spinner=False) 

1635def _wizard_trial_combos_cached(_frame, columns: tuple, participant, fingerprint: str): 

1636 """Distinct ``(participant_id, trial_id)`` pairs in one raw frame.""" 

1637 trial = trial_id_series(_frame, list(columns)).astype(str) 

1638 reader = ( 

1639 _frame[participant].astype(str) 

1640 if participant 

1641 else pd.Series([""] * len(_frame), index=_frame.index) 

1642 ) 

1643 out = pd.DataFrame({"participant_id": reader, "trial_id": trial}) 

1644 return out.drop_duplicates().reset_index(drop=True) 

1645 

1646 

1647@st.cache_data(show_spinner=False) 

1648def _wizard_reader_ids_cached(_frame, column: str, fingerprint: str) -> list: 

1649 """Distinct reader ids in one raw frame's mapped participant column. 

1650 

1651 Underscore-prefixed frame + an explicit `frame_fingerprint` key, the house 

1652 convention — see `app._cached_participant_ids`. 

1653 """ 

1654 return sorted({str(value) for value in _frame[column].dropna().unique()}) 

1655 

1656 

1657def _wizard_text_ids(raw_words, word_schema, raw_fix, fix_schema) -> list: 

1658 """The text ids the finished dataset will have, read from the raw tables. 

1659 

1660 The third-grain sibling of :func:`_wizard_reader_ids` — same reasoning: 

1661 runs before normalization, so the join report is built off the raw 

1662 frames through the mapping the user has just made, gated on a text table 

1663 actually being attached, and cached per frame. 

1664 """ 

1665 from scanpath_studio import metadata as metadata_mod 

1666 

1667 if st.session_state.get(metadata_mod.TEXT_RAW_SESSION_KEY) is None: 

1668 return [] 

1669 found: set = set() 

1670 for frame, schema in ((raw_fix, fix_schema), (raw_words, word_schema)): 

1671 column = (schema or {}).get("text_id") 

1672 if not column or frame is None or frame.empty or column not in frame.columns: 

1673 continue 

1674 found |= set(_wizard_reader_ids_cached(frame, column, frame_fingerprint(frame))) 

1675 return sorted(found) 

1676 

1677 

1678def _row_body(host): 

1679 """Indent to where the field-mapping pickers start (`_ID_ROW1_W`'s name 

1680 column), for a row that has no name of its own — the "Extra fields to 

1681 keep" picker and the "Aggregate character AOIs" toggle both describe the 

1682 table above them rather than naming a new one, so they line up under the 

1683 pickers rather than under the row-name label.""" 

1684 _, body = host.columns([_ID_ROW1_W[0], 1 - _ID_ROW1_W[0]], gap="small") 

1685 return body 

1686 

1687 

1688#: The table each upload row's prefix reads. 

1689_PREFIX_KIND = {"col_map_words": "words", "col_map_fix": "fixations"} 

1690 

1691 

1692def _wizard_table_keep_picker( 

1693 host, raw, schema, registry, prefix: str, *, noun: str 

1694) -> tuple[set, list]: 

1695 """One table's "fields to keep" decision (UX-114) — directly under that 

1696 table's own mapping, replacing the old cross-table pair of pickers 

1697 ("Filter trials by" / "Additional fields to keep") that used to be their 

1698 own wizard stage. Everything not mapped above and not kept here is 

1699 dropped at normalization; everything kept becomes available to filter 

1700 trials by, sort by, color by, or show as an info chip later — one 

1701 decision instead of two. 

1702 

1703 Returns ``(kept_source_columns, meta_dest_fields)``: the first feeds 

1704 ``compute_keep_columns(keep_columns=…)`` for this table; the second is 

1705 this table's contribution to the cross-table trial-filter condition list 

1706 (a source column detected as a trial-level "meta" condition, e.g. 

1707 Hunting/Gathering or difficulty — kept by default, same as before). 

1708 """ 

1709 if raw is None or raw.empty or schema is None: 

1710 return set(), [] 

1711 # PERF-6: categorize against the file's HEADER, not the frame — the frame 

1712 # holds only the columns the plan parsed, and the whole job of the picker 

1713 # below is to offer the ones it didn't. Naming one here adds it to the 

1714 # plan, and the file is read again under it. 

1715 header = app._uploaded_header(prefix) 

1716 source = pd.DataFrame(columns=header) if header else raw 

1717 cats = _c_categorize_columns( 

1718 source, 

1719 schema, 

1720 registry, 

1721 tuple(header) or frame_fingerprint(raw), 

1722 (prefix, _schema_key(schema)), 

1723 ) 

1724 detected = cats["detected_optional"] 

1725 unclaimed = cats["unclaimed"] 

1726 if not detected and not unclaimed: 

1727 return set(), [] 

1728 host = _row_body(host) 

1729 # #374 F13: a column already mapped (TRIAL_INDEX as the Trial ID) is never 

1730 # pre-kept, and an unmapped one that reads as a condition or an item id is. 

1731 mapped = set(cats["mapped"]) 

1732 sample = app.upload_sample(prefix, _PREFIX_KIND.get(prefix)) 

1733 if sample.empty: 

1734 sample = raw 

1735 

1736 opts: list = [] 

1737 labels: dict = {} 

1738 default: list = [] 

1739 meta_dest_by_source: dict = {} 

1740 for d in detected: 

1741 src = d["source"] 

1742 opts.append(src) 

1743 # UX-121: no more "· meta"/"· extra" suffix — each table's picker is 

1744 # already its own, so the category badge that used to help tell 

1745 # cross-table entries apart is redundant now; the field name alone. 

1746 # DATA-66: the file's own name — this screen used to show the canonical 

1747 # one (`reduced_pos` for `Reduced_POS`) before anything was normalized. 

1748 labels[src] = src 

1749 # Trial-level conditions and detected measures/linguistic features 

1750 # were both auto-kept before UX-114 split them into two pickers — 

1751 # same net defaults, offered as one choice now. 

1752 # AN-32: not the leftover measures — the ones the app uses are mapped on 

1753 # the *Reading measures* lines above, and pre-keeping the rest (last-run 

1754 # dwell, trial dwell/count, …) only widened every table by default. 

1755 if d["category"] in ("meta", "linguistic") and src not in mapped: 

1756 default.append(src) 

1757 if d["category"] == "meta": 

1758 meta_dest_by_source[src] = d["dest"] 

1759 for col in unclaimed: 

1760 opts.append(col) 

1761 labels.setdefault(col, col) 

1762 if col in sample.columns and looks_like_condition(sample[col]): 

1763 default.append(col) 

1764 

1765 key = f"wizard_keep_{prefix}" 

1766 if key not in st.session_state: 

1767 st.session_state[key] = list(default) 

1768 _clean_multiselect_state(key, opts) 

1769 

1770 inline_field_label( 

1771 host, 

1772 f"Extra fields to keep — {noun}", 

1773 "Columns not used in the mapping above. Keep the ones you want " 

1774 "available later — to filter trials by, sort by, color by, or show " 

1775 "as an info chip. Anything left out here is dropped when the " 

1776 "dataset is added.", 

1777 ) 

1778 # ENG-49: the bulk-select buttons that used to take 28% of this row are 

1779 # gone — Streamlit 1.63 puts "Select all" inside the dropdown itself 

1780 # (`select_all`, on by default under 1000 options) and has always drawn a ✕ 

1781 # to clear, so the pair had become a second copy of controls the widget now 

1782 # carries. The picker gets the whole row back. 

1783 # UX-148: `wrap=True` — directly in a column, Streamlit keeps the chips on 

1784 # one row that scrolls sideways, which hid most of a 14-field default. 

1785 chosen = set( 

1786 host.multiselect( 

1787 f"Extra fields to keep — {noun}", 

1788 options=opts, 

1789 format_func=lambda s: labels.get(s, s), 

1790 key=key, 

1791 label_visibility="collapsed", 

1792 wrap=True, 

1793 ) 

1794 ) 

1795 warning = wide_frame_warning(len(chosen), len(raw)) 

1796 if warning: 

1797 host.warning(warning) 

1798 meta_fields = [meta_dest_by_source[s] for s in chosen if s in meta_dest_by_source] 

1799 return chosen, meta_fields 

1800 

1801 

1802def _wizard_restore_config(host) -> None: 

1803 """Step 1 of the wizard: optionally restore a previously saved setup, seeding 

1804 the column mapping + kept-field choices so the user skips re-mapping. Applied 

1805 once per uploaded file; reruns so the mapping widgets pick up the values.""" 

1806 uploaded = host.file_uploader( 

1807 "Restore a saved setup (optional)", 

1808 type=["json"], 

1809 key="wizard_config_restore", 

1810 help="Re-apply a column mapping + field choices you saved earlier " 

1811 "(⬇️ Download setup file at the foot of this page).", 

1812 max_upload_size=upload_limit_mb(), 

1813 ) 

1814 if uploaded is None: 

1815 # Cleared: forget the last file, so choosing it again re-applies it. 

1816 st.session_state.pop("_wizard_config_last", None) 

1817 return 

1818 # Once per upload, not per name + size — a revised file of the same length 

1819 # is a different file (`upload_identity`). 

1820 signature = upload_identity(uploaded) 

1821 if st.session_state.get("_wizard_config_last") == signature: 

1822 return 

1823 st.session_state["_wizard_config_last"] = signature 

1824 try: 

1825 config = json.loads(uploaded.getvalue().decode("utf-8")) 

1826 except (ValueError, UnicodeDecodeError): 

1827 host.warning("That file isn't a saved setup (not valid JSON).") 

1828 return 

1829 _setup_sections = ( 

1830 "data_source", 

1831 "column_mapping", 

1832 "experimental_setup", 

1833 "filename_derive", 

1834 "keep_and_filter", 

1835 ) 

1836 if not isinstance(config, dict) or not any(k in config for k in _setup_sections): 

1837 # #374: an unrelated JSON used to toast "Restored" with nothing restored. 

1838 host.warning("That file holds no saved setup, so nothing was restored.") 

1839 return 

1840 if isinstance(config, dict): 

1841 # Overwrite: the wizard's mapping widgets were already created on a prior 

1842 # render, so their keys exist — setdefault would no-op and the restore 

1843 # would silently fail. This step runs before the widgets re-instantiate 

1844 # this pass, so writing the keys is safe, and it reruns afterwards. 

1845 _seed_column_mapping( 

1846 config.get("column_mapping"), 

1847 overwrite=True, 

1848 dataset=WIZARD_MAPPING_DATASET, 

1849 ) 

1850 # Remember the restored config's provenance so the caller can show which 

1851 # dataset (and when) it was exported from, below the upload box (9.1). 

1852 st.session_state["_wizard_restored_meta"] = { 

1853 "data_source": config.get("data_source"), 

1854 "exported_at": config.get("exported_at"), 

1855 } 

1856 # DATA-22 decision (a): a restored setup file pre-answers the 

1857 # Recording-setup step. The section is optional — a file written before 

1858 # this existed simply leaves the step unanswered, which is the honest 

1859 # outcome rather than a silent default. Additive, so per ENG-11 the 

1860 # PLOT_CONFIG_SCHEMA stays where it is. 

1861 # UX-113 Phase 3: filename/column-derive settings + keep/filter-field 

1862 # choices weren't captured before — a restored setup silently dropped 

1863 # them, forcing a re-do even though the mapping itself round-tripped. 

1864 # Both sections are optional (older files simply lack them), so no 

1865 # PLOT_CONFIG_SCHEMA bump — same precedent as experimental_setup. 

1866 if isinstance(config.get("filename_derive"), dict): 

1867 fd = config["filename_derive"] 

1868 # UX-129: a setup saved before multiple lines existed wrote one 

1869 # dict; the current writer always writes a list (one entry per 

1870 # line) — accept either, `_wizard_filename_derive` normalizes. 

1871 if isinstance(fd.get("applied"), (dict, list)): 

1872 st.session_state[_FILENAME_DERIVE_APPLIED_KEY] = fd["applied"] 

1873 widgets = fd.get("widgets") 

1874 if isinstance(widgets, dict): 

1875 for key, value in widgets.items(): 

1876 if value is not None: 

1877 st.session_state[key] = value 

1878 if isinstance(config.get("keep_and_filter"), dict): 

1879 kf = config["keep_and_filter"] 

1880 # UX-114: the per-table picks are the real source of truth — each 

1881 # `wizard_keep_<prefix>` widget re-derives `wizard_filter_fields` 

1882 # itself once the mapping resolves, so seeding those (rather than 

1883 # the flat legacy keys) is what actually reproduces the setup. 

1884 by_table = kf.get("wizard_keep_by_table") 

1885 if isinstance(by_table, dict): 

1886 for prefix, cols in by_table.items(): 

1887 if isinstance(cols, list): 

1888 st.session_state[f"wizard_keep_{prefix}"] = list(cols) 

1889 elif kf.get("wizard_keep_extra") is not None: 

1890 # A setup saved before UX-114 only has the flat cross-table 

1891 # list — apply it to both tables; each one's picker prunes 

1892 # away whatever it doesn't actually offer. 

1893 cols = list(kf["wizard_keep_extra"]) 

1894 for prefix in ("col_map_words", "col_map_fix", "col_map_raw_gaze"): 

1895 st.session_state[f"wizard_keep_{prefix}"] = list(cols) 

1896 if isinstance(config.get("experimental_setup"), dict): 

1897 # The canvas is carried in a *sibling* section by the plot-config 

1898 # writer (`tabs._build_studio_config` puts it under `canvas_px`), so 

1899 # it has to be merged in here — see `_restored_setup_snapshot`, which 

1900 # would otherwise fall back to the 2560x1440 class default and, if 

1901 # the file's provenance said "measured", pre-answer the step with a 

1902 # measured monitor nobody ever measured. 

1903 restored = dict(config["experimental_setup"]) 

1904 canvas = config.get("canvas_px") 

1905 if isinstance(canvas, dict): 

1906 for key, source in ( 

1907 ("canvas_width", "width"), 

1908 ("canvas_height", "height"), 

1909 ): 

1910 if canvas.get(source) is not None: 

1911 restored.setdefault(key, canvas[source]) 

1912 st.session_state["_wizard_restored_setup"] = restored 

1913 st.session_state.pop("_wizard_setup_restored_applied", None) 

1914 st.toast("Restored the saved setup — review it below.", icon=ICONS["success"]) 

1915 st.rerun() 

1916 

1917 

1918def _render_restored_config_caption(host) -> None: 

1919 """Below the restore box: name the dataset the restored setup came from (and 

1920 when it was exported), so the user can confirm they loaded the right one.""" 

1921 meta = st.session_state.get("_wizard_restored_meta") 

1922 if not meta: 

1923 return 

1924 source = meta.get("data_source") 

1925 exported = meta.get("exported_at") 

1926 bits = [] 

1927 if source: 

1928 bits.append(f"from **{source}**") 

1929 if exported: 

1930 try: 

1931 from datetime import datetime 

1932 

1933 bits.append(f"exported {datetime.fromisoformat(exported):%Y-%m-%d %H:%M}") 

1934 except (ValueError, TypeError): 

1935 bits.append(f"exported {exported}") 

1936 detail = " · ".join(bits) if bits else "from a saved file" 

1937 host.caption(f"✓ Restored setup {detail} — review the mapping below.") 

1938 

1939 

1940def _filename_derive_section() -> dict | None: 

1941 """The ``filename_derive`` section for a saved config (UX-113 Phase 3): the 

1942 committed derivation (``_FILENAME_DERIVE_APPLIED_KEY``) plus the live 

1943 widget values needed to redisplay the control on restore. ``None`` when 

1944 nothing has been applied — an absent section is the honest "not used". 

1945 

1946 UX-129: ``applied`` is a *list* now (one dict per derivation line), but 

1947 only line 0's widgets are captured for redisplay — a restored setup with 

1948 several lines still re-derives every one of them (the committed list 

1949 replays in full), it just starts back at one visible line if the control 

1950 is reopened to edit it, the same as a freshly opened wizard would. 

1951 """ 

1952 applied = st.session_state.get(_FILENAME_DERIVE_APPLIED_KEY) 

1953 if not applied: 

1954 return None 

1955 return { 

1956 "applied": applied, 

1957 "widgets": { 

1958 "wizard_filename_split": st.session_state.get("wizard_filename_split"), 

1959 "wizard_filename_table": st.session_state.get("wizard_filename_table"), 

1960 "wizard_filename_column": st.session_state.get("wizard_filename_column"), 

1961 "wizard_filename_mode": st.session_state.get("wizard_filename_mode"), 

1962 "wizard_filename_delim": st.session_state.get("wizard_filename_delim"), 

1963 "wizard_filename_regex": st.session_state.get("wizard_filename_regex"), 

1964 "wizard_filename_regex_lower": st.session_state.get( 

1965 "wizard_filename_regex_lower" 

1966 ), 

1967 }, 

1968 } 

1969 

1970 

1971def _keep_and_filter_section() -> dict | None: 

1972 """The ``keep_and_filter`` section for a saved config (UX-113 Phase 3, 

1973 reshaped per-table by UX-114). ``None`` when nothing was touched. 

1974 

1975 UX-114 replaced the two cross-table pickers (``wizard_filter_by`` / 

1976 ``wizard_keep_extra``) with one per-table multiselect each 

1977 (``wizard_keep_<prefix>``) — a **new, additively-named** key, 

1978 ``wizard_keep_by_table``, carries those; the old flat keys are still 

1979 written too (kept in sync from the per-table picks) so a setup saved here 

1980 still restores cleanly through any older reader of this section. No 

1981 ``PLOT_CONFIG_SCHEMA`` bump — this whole section is already optional and 

1982 read defensively. 

1983 """ 

1984 by_table = { 

1985 # UX-120: raw gaze joined the two tables with their own keep-picker. 

1986 prefix: list(st.session_state[key]) 

1987 for prefix in ("col_map_words", "col_map_fix", "col_map_raw_gaze") 

1988 if (key := f"wizard_keep_{prefix}") in st.session_state 

1989 and st.session_state[key] 

1990 } 

1991 filter_fields = st.session_state.get("wizard_filter_fields") 

1992 if not by_table and not filter_fields: 

1993 return None 

1994 return { 

1995 "wizard_keep_by_table": by_table or None, 

1996 "wizard_filter_by": filter_fields, 

1997 "wizard_keep_extra": sorted({c for cols in by_table.values() for c in cols}) 

1998 or None, 

1999 } 

2000 

2001 

2002def _wizard_setup_config() -> dict: 

2003 """The current wizard setup as a JSON-able dict: the column mapping plus 

2004 provenance, in the schema the restore step (``_wizard_restore_config``) reads 

2005 back. Lets a user save their mapping and re-apply it to similar data later.""" 

2006 from datetime import datetime 

2007 

2008 from scanpath_studio import __version__ 

2009 

2010 return { 

2011 # The settings file's format — this setup file can be loaded through 

2012 # 🔗 Share → File's uploader too (which applies its setup, not its 

2013 # mapping, since UX-179), so it stamps the same single-source-of-truth 

2014 # schema version (ENG-11) rather than a literal. 

2015 "schema": PLOT_CONFIG_SCHEMA, 

2016 "app": {"name": "Scanpath Studio", "version": __version__}, 

2017 "exported_at": datetime.now().isoformat(timespec="seconds"), 

2018 "data_source": (st.session_state.get("wizard_dataset_name") or "").strip() 

2019 or None, 

2020 "column_mapping": _collect_column_mapping(), 

2021 # DATA-22 §7 surface 3: the recording setup + its provenance, so a 

2022 # re-applied setup file carries "the monitor was assumed" rather than 

2023 # quietly re-deriving a number the recipient would read as measured. 

2024 "experimental_setup": current_setup_section(), 

2025 "filename_derive": _filename_derive_section(), 

2026 "keep_and_filter": _keep_and_filter_section(), 

2027 } 

2028 

2029 

2030def current_setup_section() -> dict | None: 

2031 """The ``experimental_setup`` section for a saved config, or ``None``. 

2032 

2033 Shared by ``_wizard_setup_config`` and ``tabs._build_studio_config`` so both 

2034 writers emit the same shape. Returns ``None`` when the wizard has not 

2035 resolved a snapshot yet — an absent section reads as "unknown", which is 

2036 what it is; an all-defaults one would read as an answer. 

2037 """ 

2038 payload = st.session_state.get("_wizard_setup_snapshot") 

2039 return dict(payload) if isinstance(payload, dict) else None 

2040 

2041 

2042def _render_setup_download(host) -> None: 

2043 """Export the current column mapping as a JSON setup file, so it can be 

2044 re-applied later via the wizard's *Restore a saved setup* step. Rendered 

2045 beside the **Add dataset** button (UX-53).""" 

2046 host.download_button( 

2047 # UX-93: short enough to sit on ONE line at the ✕ Cancel width the 

2048 # footer now uses — "Download setup (JSON)" wrapped to two, making the 

2049 # pair 55 px and 40 px tall side by side. What it saves and how to load 

2050 # it back is on the tooltip, where the sentence was already. 

2051 f"{ICONS['download']} Download setup file", 

2052 data=json.dumps(_wizard_setup_config(), indent=2), 

2053 file_name="scanpath_studio_setup.json", 

2054 mime="application/json", 

2055 key="wizard_setup_download", 

2056 width="stretch", 

2057 help="Save this column mapping to re-use on similar data — restore it " 

2058 "from *↩️ Restore a saved setup* beside *Upload data tables*.", 

2059 ) 

2060 

2061 

2062#: UX-93 — the wizard's footer row. `1.4` is the ✕ Cancel width from the sticky 

2063#: bar's `[7.2, 2.2, 1.4]`, so the two things you do with a finished setup are 

2064#: the same size as the way out of it, side by side, with the rest of the line 

2065#: left empty. They used to be either two full-width banners stacked (the 

2066#: blocked path) or a 50/50 split of the whole section (the finished path) — 

2067#: neither of which reads as "two buttons". 

2068_FOOTER_ROW_W = (1.4, 1.4, 8.0) 

2069 

2070 

2071def _wizard_footer(host, *, disabled: bool, help_text: str, on_click=None) -> None: 

2072 """⬇️ Download setup file · ✅ Add dataset, one line, matched widths (UX-93). 

2073 

2074 One place for all three endings — blocked on required fields, blocked on a 

2075 mapping the pipeline rejected, and finished — which is what keeps them the 

2076 same size and the same colour. ✅ Add dataset is `primary` on every path: 

2077 it is the page's one commit, and #UX-66 r2 already paired ✕ Cancel with it 

2078 in the same filled blue, as the two ends of one decision. 

2079 

2080 UX-113: the row is keyed so `styles.py` can widen the two button columns 

2081 (and shrink the empty `_rest` one) on a narrow screen — `_FOOTER_ROW_W`'s 

2082 compact desktop width leaves each button too few pixels there, and "Save 

2083 setup" / "Add dataset" wrap to two lines. 

2084 

2085 UX-113 also gave it a real `st.divider()` above — heavier and more spaced 

2086 (its own keyed container, `styles.py` widens the margin on both sides) 

2087 than the mapping blocks' own hairline (`.sps-wiz-blockgap`) on purpose: 

2088 this is the one commit for the *whole* wizard, not the end of stage 5, and 

2089 it should read as a clear break rather than one more row under Recording 

2090 setup. 

2091 """ 

2092 divider_host = host.container(key="wizard_footer_divider") 

2093 divider_host.divider() 

2094 row = host.container(key="wizard_footer_row") 

2095 save_col, add_col, _rest = row.columns( 

2096 _FOOTER_ROW_W, gap="small", vertical_alignment="center" 

2097 ) 

2098 _render_setup_download(save_col) 

2099 add_col.button( 

2100 f"{ICONS['confirm']} Add dataset", 

2101 type="primary", 

2102 key="wizard_finalize", 

2103 disabled=disabled, 

2104 on_click=on_click, 

2105 width="stretch", 

2106 help=help_text, 

2107 ) 

2108 

2109 

2110# ----------------------------------------------------------------------------- 

2111# Step 4 · Recording setup (DATA-22 §3) 

2112# 

2113# The old wizard rendered this panel as step *2* — before the upload — and 

2114# silently seeded a 2560x1440 monitor, 597 mm, 800 mm and a 16 px font, with 

2115# nothing telling the user those were guesses. Estimating from the data was not 

2116# even possible there, because there was no data yet. 

2117# 

2118# Now it sits after the upload and asks, per group, *how do you know?* — with 

2119# `index=None` so nothing is preselected. The user either knows the values, 

2120# derives them from their own data, knowingly takes a named default, or (for 

2121# visual-angle units only) skips. The answer is recorded as a `Provenance` that 

2122# travels with the dataset, so a reader downstream can tell a measured screen 

2123# from an assumed one. 

2124# 

2125# The gate is deliberately hard: **Add dataset** stays disabled until all three 

2126# are answered. Nobody can be stranded by it — *Estimate from my data* always 

2127# exists for the screen group and always succeeds — but nobody gets a silent 

2128# default either. 

2129# ----------------------------------------------------------------------------- 

2130 

2131_SETUP_MODE_KEYS = {g: f"wizard_setup_{g}_mode" for g in SETUP_GROUPS} 

2132 

2133_SCREEN_KNOW = "I know the resolution" 

2134_SCREEN_ESTIMATE = "Estimate from my data" 

2135_SCREEN_DEFAULT = "Use a common default (2560×1440)" 

2136 

2137_GEOM_KNOW = "I know them" 

2138#: The setup headings as drawn (UX-58's short forms) — what blockers name. 

2139_SETUP_HEADINGS = {"screen": "Screen", "geometry": "Physical size", "text": "Text size"} 

2140_GEOM_DEFAULT = "Use typical lab values (screen 597 mm wide, viewed from 800 mm)" 

2141_GEOM_SKIP = "Skip — I don't need visual-angle units" 

2142 

2143_TEXT_BOXES = "Scale to the word boxes" 

2144_TEXT_FONT = "I know the stimulus font" 

2145_TEXT_DEFAULT = "Use a default (16 px)" 

2146 

2147_SETUP_PROVENANCE = { 

2148 _SCREEN_KNOW: Provenance.MEASURED, 

2149 _SCREEN_ESTIMATE: Provenance.ESTIMATED, 

2150 _SCREEN_DEFAULT: Provenance.ASSUMED, 

2151 _GEOM_KNOW: Provenance.MEASURED, 

2152 _GEOM_DEFAULT: Provenance.ASSUMED, 

2153 _GEOM_SKIP: Provenance.SKIPPED, 

2154 _TEXT_BOXES: Provenance.MEASURED, 

2155 _TEXT_FONT: Provenance.MEASURED, 

2156 _TEXT_DEFAULT: Provenance.ASSUMED, 

2157} 

2158 

2159 

2160def _recalled(field: str, fallback): 

2161 """A value remembered from the previous dataset set up this session. 

2162 

2163 Decision (d): answers are recalled across datasets as *pre-filled values with 

2164 the radio reset* — fast for a second export from the same lab, but the user 

2165 still has to assert that the setup applies to this dataset too. 

2166 """ 

2167 return (st.session_state.get("_wizard_setup_recall") or {}).get(field, fallback) 

2168 

2169 

2170def _remember_setup(values: dict) -> None: 

2171 """Stash this dataset's answers as pre-fill for the next one (values only).""" 

2172 recall = dict(st.session_state.get("_wizard_setup_recall") or {}) 

2173 recall.update(values) 

2174 st.session_state["_wizard_setup_recall"] = recall 

2175 

2176 

2177def _setup_mode( 

2178 host, 

2179 group: str, 

2180 options: list, 

2181 help_text: str, 

2182 label=None, 

2183 *, 

2184 key_prefix: str = "wizard", 

2185 persist: dict | None = None, 

2186): 

2187 """One setup group's radio, namespaced for add or edit. 

2188 

2189 Returns the chosen label or ``None``. The mode keys are wizard-local UI state 

2190 and deliberately **not** wire format (same reasoning as ``share_identity_mode`` 

2191 in ``url_state.py``): what travels is the resolved value plus its provenance, 

2192 not which radio button produced it. 

2193 

2194 ``label`` overrides the heading for display only (UX-58). Three groups 

2195 side by side leave no room for *Physical size & viewing distance*, and a 

2196 heading that wraps to two lines drops its column out of line with the other 

2197 two — which is the whole point of the row. `SETUP_GROUP_LABELS` stays the 

2198 name everything else reports by. 

2199 """ 

2200 # UX-90: every setup group is mandatory — *Add dataset* stays disabled until 

2201 # each says how it is known — so each carries the same trailing `*` the 

2202 # required mapping fields do. One convention for "you must answer this", 

2203 # rather than a starred column mapping above an unstarred set of questions. 

2204 return host.radio( 

2205 f"{label or SETUP_GROUP_LABELS[group]} *", 

2206 options, 

2207 index=None, 

2208 key=f"{key_prefix}_setup_{group}_mode", 

2209 help=help_text, 

2210 **(persist or {}), 

2211 ) 

2212 

2213 

2214def _wizard_setup_step( 

2215 host, 

2216 words_raw, 

2217 fix_raw, 

2218 has_boxes: bool, 

2219 *, 

2220 key_prefix: str = "wizard", 

2221 initial: SetupSnapshot | None = None, 

2222 publish: bool = True, 

2223 estimate=None, 

2224) -> SetupSnapshot: 

2225 """Render the three Recording-setup groups and resolve them to a snapshot. 

2226 

2227 ``estimate`` (DATA-46) is a zero-argument callable giving the *Estimate from 

2228 my data* size; when it is ``None``, ``words_raw`` / ``fix_raw`` must already 

2229 carry canonical coordinates (the editor's stored frames) and are measured 

2230 directly. Either way it is called only when that answer is chosen. 

2231 

2232 Writes the resolved values into the existing ``global_*`` wire-format keys 

2233 (unchanged — the *values* were always wire format; only the provenance is 

2234 new). Safe to write here because the wizard owns the page: ``app.main`` 

2235 returns before the rail renders, so no widget on a ``global_*`` key exists 

2236 this run. 

2237 """ 

2238 # ✏️ Edit dataset lives on the Data page, which runs only while it is the 

2239 # open view: without this a trip to the Scanpath view mid-edit dropped the 

2240 # setup draft (Streamlit forgets an unrendered widget's key) while the 

2241 # mapping fields beside it — `persist_state` since DATA-26 — kept theirs. 

2242 persist = {"persist_state": "session"} if initial is not None else {} 

2243 # UX-58: three columns, one per group, so their headings sit at the same 

2244 # line height. Each column starts with its own radio, which is what keeps 

2245 # them level even though what follows differs per answer (two number inputs, 

2246 # an info box, or a caption) and so the columns end at different heights. 

2247 # The description that used to print here is now the section's hover text. 

2248 screen_host, geom_host, text_host = host.columns(3, gap="medium") 

2249 

2250 # The Data Management editor reuses this exact control layout. Seed its 

2251 # three mode choices from the saved snapshot once; the add flow keeps its 

2252 # deliberate unanswered state because ``initial`` is None. 

2253 if initial is not None: 

2254 screen_modes = { 

2255 Provenance.MEASURED: _SCREEN_KNOW, 

2256 Provenance.ESTIMATED: _SCREEN_ESTIMATE, 

2257 Provenance.ASSUMED: _SCREEN_DEFAULT, 

2258 } 

2259 geometry_modes = { 

2260 Provenance.MEASURED: _GEOM_KNOW, 

2261 Provenance.ASSUMED: _GEOM_DEFAULT, 

2262 Provenance.SKIPPED: _GEOM_SKIP, 

2263 } 

2264 text_mode = ( 

2265 _TEXT_BOXES 

2266 if initial.scale_text_to_boxes and has_boxes 

2267 else _TEXT_DEFAULT 

2268 if initial.text_provenance is Provenance.ASSUMED 

2269 else _TEXT_FONT 

2270 ) 

2271 st.session_state.setdefault( 

2272 f"{key_prefix}_setup_screen_mode", 

2273 screen_modes.get(initial.screen_provenance, _SCREEN_KNOW), 

2274 ) 

2275 st.session_state.setdefault( 

2276 f"{key_prefix}_setup_geometry_mode", 

2277 geometry_modes.get(initial.geometry_provenance, _GEOM_DEFAULT), 

2278 ) 

2279 st.session_state.setdefault(f"{key_prefix}_setup_text_mode", text_mode) 

2280 

2281 # --- Screen ------------------------------------------------------------- 

2282 screen_mode = _setup_mode( 

2283 screen_host, 

2284 "screen", 

2285 [_SCREEN_KNOW, _SCREEN_ESTIMATE, _SCREEN_DEFAULT], 

2286 "The presentation monitor's resolution in pixels. Everything is drawn in " 

2287 "these coordinates.", 

2288 label="Screen", 

2289 key_prefix=key_prefix, 

2290 persist=persist, 

2291 ) 

2292 canvas_w = ( 

2293 initial.canvas_width if initial is not None else _recalled("canvas_width", 2560) 

2294 ) 

2295 canvas_h = ( 

2296 initial.canvas_height 

2297 if initial is not None 

2298 else _recalled("canvas_height", 1440) 

2299 ) 

2300 if screen_mode == _SCREEN_KNOW: 

2301 w_col, h_col = screen_host.columns(2, gap="small") 

2302 canvas_w = w_col.number_input( 

2303 "Width (px)", 

2304 100, 

2305 10000, 

2306 int(canvas_w), 

2307 key=f"{key_prefix}_setup_screen_w", 

2308 **persist, 

2309 ) 

2310 canvas_h = h_col.number_input( 

2311 "Height (px)", 

2312 100, 

2313 10000, 

2314 int(canvas_h), 

2315 key=f"{key_prefix}_setup_screen_h", 

2316 **persist, 

2317 ) 

2318 elif screen_mode == _SCREEN_ESTIMATE: 

2319 est_w, est_h = ( 

2320 estimate() 

2321 if estimate is not None 

2322 else compute_canvas_size(words_raw, fix_raw) 

2323 ) 

2324 # DATA-46: on ✏️ Edit dataset, a screen that was *saved* as an estimate 

2325 # keeps the size it was saved with. Re-estimating on every open meant a 

2326 # ✅ Save changes with nothing touched rewrote the canvas of every figure 

2327 # from this dataset — the editor must not change what the user did not. 

2328 # A fresh estimate is still one click away, and says what it would be. 

2329 reestimate_key = f"{key_prefix}_setup_reestimate" 

2330 keep_saved = ( 

2331 initial is not None 

2332 and initial.screen_provenance is Provenance.ESTIMATED 

2333 and not st.session_state.get(reestimate_key) 

2334 ) 

2335 if keep_saved: 

2336 canvas_w, canvas_h = int(initial.canvas_width), int(initial.canvas_height) 

2337 screen_host.info( 

2338 f"Estimated **{canvas_w} × {canvas_h} px** when this dataset was " 

2339 "added — a **lower bound** from the extent of its word boxes and " 

2340 "fixations." 

2341 ) 

2342 if (est_w, est_h) != (canvas_w, canvas_h): 

2343 screen_host.button( 

2344 f"↻ Use the current estimate ({est_w} × {est_h} px)", 

2345 key=f"{key_prefix}_setup_reestimate_btn", 

2346 on_click=lambda: st.session_state.__setitem__(reestimate_key, True), 

2347 help="Re-estimate the screen from this dataset's data as " 

2348 "mapped above. Nothing changes until you save.", 

2349 ) 

2350 else: 

2351 canvas_w, canvas_h = est_w, est_h 

2352 screen_host.info( 

2353 f"Estimated **{est_w} × {est_h} px** from the extent of your word " 

2354 "boxes and fixations. This is a **lower bound** — text rarely fills " 

2355 "the whole screen, so the real monitor was probably larger." 

2356 ) 

2357 elif screen_mode == _SCREEN_DEFAULT: 

2358 canvas_w, canvas_h = 2560, 1440 

2359 screen_host.caption("Recorded as **assumed** — a common 1440p monitor.") 

2360 

2361 # --- Physical size & viewing distance ----------------------------------- 

2362 geom_mode = _setup_mode( 

2363 geom_host, 

2364 "geometry", 

2365 [_GEOM_KNOW, _GEOM_DEFAULT, _GEOM_SKIP], 

2366 "Needed only to express distances in degrees of visual angle. Skipping is " 

2367 "a real answer — the app then hides the numbers it cannot honestly derive.", 

2368 label="Physical size", 

2369 key_prefix=key_prefix, 

2370 persist=persist, 

2371 ) 

2372 mon_mm = float( 

2373 initial.monitor_width_mm 

2374 if initial is not None 

2375 else _recalled("monitor_width_mm", 597.0) 

2376 ) 

2377 dist_mm = float( 

2378 initial.viewing_distance_mm 

2379 if initial is not None 

2380 else _recalled("viewing_distance_mm", 800.0) 

2381 ) 

2382 if geom_mode == _GEOM_KNOW: 

2383 mon_mm = geom_host.number_input( 

2384 "Monitor width (mm)", 

2385 50.0, 

2386 2000.0, 

2387 mon_mm, 

2388 key=f"{key_prefix}_setup_monitor_mm", 

2389 **persist, 

2390 ) 

2391 dist_mm = geom_host.number_input( 

2392 "Viewing distance (mm)", 

2393 50.0, 

2394 5000.0, 

2395 dist_mm, 

2396 key=f"{key_prefix}_setup_distance_mm", 

2397 **persist, 

2398 ) 

2399 elif geom_mode == _GEOM_DEFAULT: 

2400 mon_mm, dist_mm = 597.0, 800.0 

2401 geom_host.caption("Recorded as **assumed** — typical lab values.") 

2402 elif geom_mode == _GEOM_SKIP: 

2403 geom_host.caption( 

2404 "Visual-angle units stay **hidden** for this dataset rather than being " 

2405 "computed from a default." 

2406 ) 

2407 

2408 # --- Reading text size --------------------------------------------------- 

2409 text_options = [_TEXT_FONT, _TEXT_DEFAULT] 

2410 if has_boxes: 

2411 # Only offered when there are boxes to scale to — otherwise it is an 

2412 # option that silently does nothing. 

2413 text_options.insert(0, _TEXT_BOXES) 

2414 text_mode = _setup_mode( 

2415 text_host, 

2416 "text", 

2417 text_options, 

2418 "How big the reading text was drawn. Word labels are rendered at this size " 

2419 "so the figure matches what the participant saw.", 

2420 label="Text size", 

2421 key_prefix=key_prefix, 

2422 persist=persist, 

2423 ) 

2424 scale_to_boxes = True 

2425 base_font = int( 

2426 initial.base_font_size 

2427 if initial is not None 

2428 else _recalled("base_font_size", 16) 

2429 ) 

2430 font_family = str( 

2431 initial.font_family 

2432 if initial is not None 

2433 else _recalled("font_family", FONT_FAMILY) 

2434 ) 

2435 if text_mode == _TEXT_BOXES: 

2436 scale_to_boxes = True 

2437 text_host.caption( 

2438 "Label size is derived from each word box — the usual choice." 

2439 ) 

2440 elif text_mode == _TEXT_FONT: 

2441 scale_to_boxes = False 

2442 initial_font_pt = float(_recalled("stimulus_font_pt", 12.0)) 

2443 if initial is not None and mon_mm > 0 and geom_mode != _GEOM_SKIP: 

2444 initial_dpi = float(canvas_w) / (float(mon_mm) / 25.4) 

2445 initial_font_pt = float(initial.base_font_size) * 72.0 / initial_dpi 

2446 font_pt = text_host.number_input( 

2447 "Stimulus font (pt)", 

2448 4.0, 

2449 96.0, 

2450 initial_font_pt, 

2451 key=f"{key_prefix}_setup_font_pt", 

2452 **persist, 

2453 ) 

2454 font_family = text_host.text_input( 

2455 "Font family", 

2456 value=font_family, 

2457 key=f"{key_prefix}_setup_font_family", 

2458 **persist, 

2459 ) 

2460 # pt→px needs a DPI, which needs the physical width. Under a skipped 

2461 # geometry group there is no honest DPI, so the conversion is withheld 

2462 # and the point size falls back to being read as pixels. 

2463 if geom_mode == _GEOM_SKIP: 

2464 text_host.warning( 

2465 "Converting points to pixels needs the monitor width, skipped " 

2466 "under **Physical size**, so the size is read as **pixels**." 

2467 ) 

2468 base_font = int(min(max(round(font_pt), 6), 72)) 

2469 else: 

2470 dpi = float(canvas_w) / (float(mon_mm) / 25.4) if mon_mm > 0 else 96.0 

2471 base_font = int(min(max(round(font_pt_to_px(font_pt, dpi)), 6), 72)) 

2472 text_host.caption(f"→ **{base_font} px** at {dpi:.0f} DPI.") 

2473 elif text_mode == _TEXT_DEFAULT: 

2474 scale_to_boxes = False 

2475 base_font = 16 

2476 text_host.caption("Recorded as **assumed** — a 16 px reading font.") 

2477 

2478 snapshot = SetupSnapshot( 

2479 canvas_width=int(canvas_w), 

2480 canvas_height=int(canvas_h), 

2481 monitor_width_mm=float(mon_mm), 

2482 viewing_distance_mm=float(dist_mm), 

2483 base_font_size=int(base_font), 

2484 font_family=font_family, 

2485 line_spacing=float( 

2486 initial.line_spacing 

2487 if initial is not None 

2488 else st.session_state.get("global_line_spacing", 3.0) 

2489 ), 

2490 scale_text_to_boxes=bool(scale_to_boxes), 

2491 screen_provenance=_SETUP_PROVENANCE.get(screen_mode), 

2492 geometry_provenance=_SETUP_PROVENANCE.get(geom_mode), 

2493 text_provenance=_SETUP_PROVENANCE.get(text_mode), 

2494 ) 

2495 

2496 # Publish the snapshot for the save/restore + export writers. A partial one 

2497 # resolves to None, so `current_setup_section` writes nothing rather than an 

2498 # all-defaults section that would read as a real answer. 

2499 if publish: 

2500 st.session_state["_wizard_setup_snapshot"] = ( 

2501 snapshot.to_dict() if snapshot.is_answered() else None 

2502 ) 

2503 

2504 # Publish each group's values as soon as *that* group is answered, into the 

2505 # wire-format `global_*` keys the rest of the app reads. The hard gate is on 

2506 # **Add dataset**, not on a setting taking effect — a user who has just told 

2507 # us the resolution should see the canvas change now, not after answering two 

2508 # unrelated questions. Only the values were ever wire format; the provenance 

2509 # beside them is what is new. 

2510 recall: dict = {} 

2511 if publish and snapshot.screen_provenance is not None: 

2512 st.session_state["global_canvas_width"] = snapshot.canvas_width 

2513 st.session_state["global_canvas_height"] = snapshot.canvas_height 

2514 recall["canvas_width"] = snapshot.canvas_width 

2515 recall["canvas_height"] = snapshot.canvas_height 

2516 if publish and snapshot.geometry_provenance not in (None, Provenance.SKIPPED): 

2517 st.session_state["global_monitor_width_mm"] = snapshot.monitor_width_mm 

2518 st.session_state["global_viewing_distance_mm"] = snapshot.viewing_distance_mm 

2519 recall["monitor_width_mm"] = snapshot.monitor_width_mm 

2520 recall["viewing_distance_mm"] = snapshot.viewing_distance_mm 

2521 if publish and snapshot.text_provenance is not None: 

2522 st.session_state["global_base_font_size"] = snapshot.base_font_size 

2523 st.session_state["global_font_family"] = snapshot.font_family 

2524 st.session_state["global_scale_text_to_boxes"] = snapshot.scale_text_to_boxes 

2525 recall["base_font_size"] = snapshot.base_font_size 

2526 recall["font_family"] = snapshot.font_family 

2527 if publish and recall: 

2528 _remember_setup(recall) 

2529 

2530 # UX-90 — an error, not a warning, and only once the user has actually tried 

2531 # to add. Before that an unanswered question is one they have not reached 

2532 # yet, and saying so in yellow on arrival made the page open already 

2533 # complaining. Same rule the mapping fields follow 

2534 # (`controls.ADD_ATTEMPTED_KEY`), so one click now turns the whole page red 

2535 # at once instead of it nagging in two different tenses. 

2536 unanswered = [g for g, p in snapshot.provenance.items() if p is None] 

2537 if publish and unanswered and st.session_state.get(ADD_ATTEMPTED_KEY): 

2538 host.error( 

2539 "Still to answer: " 

2540 + ", ".join(f"**{_SETUP_HEADINGS[g]}**" for g in unanswered) 

2541 + ". Pick an answer for each, then press Add dataset again." 

2542 ) 

2543 _mark_missing_setup_groups(unanswered) 

2544 return snapshot 

2545 

2546 

2547def _mark_missing_setup_groups(unanswered: list) -> None: 

2548 """Ring the unanswered required setup radios in red (UX-90). 

2549 

2550 One ``<style>`` block for all of them, targeting each radio's `.st-key-…` 

2551 container — the same technique as `controls._emit_field_tints`, and for the 

2552 same reason: a wrapper element per group would be more DOM on a page whose 

2553 whole problem is length. 

2554 

2555 A ring and a red label rather than a fill: the group is a list of radio 

2556 options, and tinting three option rows reads as three separate problems 

2557 instead of one unanswered question. 

2558 """ 

2559 keys = [_SETUP_MODE_KEYS[group] for group in unanswered] 

2560 box = ", ".join(f".st-key-{key} > div" for key in keys) 

2561 label = ", ".join(f".st-key-{key} label p" for key in keys) 

2562 st.markdown( 

2563 "<style>" 

2564 f"{box} {{ border: 1px solid rgba(239, 68, 68, 0.85);" 

2565 " border-radius: 0.4rem; padding: 0.35rem 0.5rem;" 

2566 " background: rgba(239, 68, 68, 0.06); }" 

2567 f"{label} {{ color: rgb(239, 68, 68); }}" 

2568 "</style>", 

2569 unsafe_allow_html=True, 

2570 ) 

2571 

2572 

2573def _restored_setup_snapshot() -> SetupSnapshot | None: 

2574 """The ``experimental_setup`` section of a restored setup JSON, if any. 

2575 

2576 Decision (a): a restored setup file **does** pre-answer step 4 — it recorded 

2577 a real choice once, and re-asking would be pedantry rather than rigour. 

2578 

2579 But only for what it actually recorded. A file that carries a *screen* 

2580 provenance without a canvas cannot pre-answer the screen: `from_dict` would 

2581 fill 2560x1440 from the class default, and a `measured` badge on top of it 

2582 would be the exact silent inheritance this whole step exists to prevent. In 

2583 that case the screen group is reset to unanswered and the user is asked. 

2584 """ 

2585 payload = st.session_state.get("_wizard_restored_setup") 

2586 if not payload: 

2587 return None 

2588 return SetupSnapshot.from_dict(payload, fallback=SetupSnapshot()) 

2589 

2590 

2591def _restored_setup_answerable_groups() -> set: 

2592 """Which groups a restored file may pre-answer — those it actually recorded. 

2593 

2594 A plot config written by `tabs._build_studio_config` carries the physical 

2595 geometry and typography but keeps the canvas in a sibling `canvas_px` 

2596 section; when that is absent too, the screen size in the snapshot is the 

2597 class default rather than anything the file stated, so the screen group is 

2598 left for the user to answer. 

2599 """ 

2600 payload = st.session_state.get("_wizard_restored_setup") or {} 

2601 groups = set(SETUP_GROUPS) 

2602 if payload.get("canvas_width") is None or payload.get("canvas_height") is None: 

2603 groups.discard("screen") 

2604 return groups 

2605 

2606 

2607def _apply_restored_setup(snapshot: SetupSnapshot) -> None: 

2608 """Pre-answer the Recording-setup radios from a restored file (once).""" 

2609 if st.session_state.get("_wizard_setup_restored_applied"): 

2610 return 

2611 st.session_state["_wizard_setup_restored_applied"] = True 

2612 answerable = _restored_setup_answerable_groups() 

2613 by_prov = { 

2614 "screen": { 

2615 Provenance.MEASURED: _SCREEN_KNOW, 

2616 Provenance.ESTIMATED: _SCREEN_ESTIMATE, 

2617 Provenance.ASSUMED: _SCREEN_DEFAULT, 

2618 }, 

2619 "geometry": { 

2620 Provenance.MEASURED: _GEOM_KNOW, 

2621 Provenance.ASSUMED: _GEOM_DEFAULT, 

2622 Provenance.SKIPPED: _GEOM_SKIP, 

2623 }, 

2624 "text": { 

2625 Provenance.MEASURED: _TEXT_FONT, 

2626 Provenance.ASSUMED: _TEXT_DEFAULT, 

2627 }, 

2628 } 

2629 for group, provenance in snapshot.provenance.items(): 

2630 if group not in answerable: 

2631 continue 

2632 label = by_prov.get(group, {}).get(provenance) 

2633 if label is not None: 

2634 st.session_state[_SETUP_MODE_KEYS[group]] = label 

2635 remembered = { 

2636 "monitor_width_mm": snapshot.monitor_width_mm, 

2637 "viewing_distance_mm": snapshot.viewing_distance_mm, 

2638 "base_font_size": snapshot.base_font_size, 

2639 "font_family": snapshot.font_family, 

2640 } 

2641 # Only remember a canvas the file actually carried; otherwise this would 

2642 # write the class default into the live geometry as if it were the user's. 

2643 if "screen" in answerable: 

2644 remembered["canvas_width"] = snapshot.canvas_width 

2645 remembered["canvas_height"] = snapshot.canvas_height 

2646 _remember_setup(remembered) 

2647 

2648 

2649def _render_multipleye_upload(body, active: bool) -> _UploadResult: 

2650 """MultiplEYE preset for the Add-dataset wizard (the "MultiplEYE" format). 

2651 

2652 Skips the generic column-mapping steps: the user uploads the corpus's 

2653 scanpath/fixation CSVs (+ optional word-AOI CSVs) and the recipe 

2654 (``datasets.multipleye_frames_from_uploads``) parses participant / session / 

2655 trial / stimulus from the file names, makes each *stimulus* a trial with its 

2656 pages (and, when the question AOI + answer-layout version files are uploaded 

2657 too, its comprehension-question screens) as ordered screens inside it, 

2658 aggregates character AOIs into word boxes, and case-matches the (lowercase) 

2659 AOI file names to the (CamelCase) stimuli. Produces the same normalized frames 

2660 + ``_wizard_finalize_payload`` as a finished generic upload, so finalize / 

2661 reload behave identically.""" 

2662 from scanpath_studio.datasets import ( 

2663 MULTIPLEYE_FIX_SCHEMA, 

2664 MULTIPLEYE_MONITOR, 

2665 multipleye_frames_from_uploads, 

2666 multipleye_word_schema, 

2667 ) 

2668 

2669 if active: 

2670 body.caption( 

2671 "Upload the MultiplEYE **scanpath** (or fixation) CSVs and, for word " 

2672 "boxes, the **AOI** CSVs — identity is read from the file names, each " 

2673 "stimulus becomes a trial whose pages are screens you can step " 

2674 "through, and character AOIs are aggregated into word boxes " 

2675 "automatically." 

2676 ) 

2677 # Seed the MultiplEYE presentation monitor (true-to-scale default). 

2678 st.session_state.setdefault("global_canvas_width", MULTIPLEYE_MONITOR[0]) 

2679 st.session_state.setdefault("global_canvas_height", MULTIPLEYE_MONITOR[1]) 

2680 app.render_canvas_controls( 

2681 empty_words_frame(), 

2682 empty_fixations_frame(), 

2683 data_choice=None, 

2684 slot=body, 

2685 expanded=False, 

2686 title="Monitor, font & text scaling", 

2687 ) 

2688 

2689 fix_df = app._read_uploaded_frame( 

2690 uploader_label="Scanpath / fixation CSVs", 

2691 upload_help="The per-trial *_scanpath.csv (preferred — they carry the word " 

2692 "index) or *_fixation.csv files. Drop in as many as you like.", 

2693 state_prefix="mpe_fix", 

2694 multi=True, 

2695 container=body, 

2696 ) 

2697 aoi_df = app._read_uploaded_frame( 

2698 uploader_label="Word AOI CSVs (optional)", 

2699 upload_help="The per-stimulus *_aoi.csv files (character interest areas). " 

2700 "Optional — without them you get fixations and no word boxes. To get the " 

2701 "comprehension-question screens too, add the *_aoi_questions.csv files " 

2702 "AND stimulus_order_versions_*.csv here (the version table says which " 

2703 "answer layout each reader saw; without it the question screens are " 

2704 "skipped), and upload the *_fixation.csv files — the scanpath export " 

2705 "does not contain those screens.", 

2706 state_prefix="mpe_aoi", 

2707 multi=True, 

2708 container=body, 

2709 ) 

2710 questions_df = app._read_uploaded_frame( 

2711 uploader_label="Comprehension questions (optional)", 

2712 upload_help="The multipleye_comprehension_questions_*.xlsx workbook — " 

2713 "adds the questions to the Stimulus & context panel.", 

2714 state_prefix="mpe_questions", 

2715 multi=False, 

2716 container=body, 

2717 ) 

2718 participant_df = app._read_uploaded_frame( 

2719 uploader_label="participant_data.csv (optional)", 

2720 upload_help="Participant metadata (age / gender / languages…) → Trial Info chips.", 

2721 state_prefix="mpe_participant", 

2722 multi=False, 

2723 container=body, 

2724 ) 

2725 

2726 if fix_df.empty: 

2727 if active: 

2728 body.info( 

2729 f"{ICONS['upload']} Upload MultiplEYE scanpath / fixation CSVs to begin." 

2730 ) 

2731 return _UploadResult( 

2732 empty_words_frame(), 

2733 empty_fixations_frame(), 

2734 pd.DataFrame(), 

2735 pd.DataFrame(), 

2736 pd.DataFrame(), 

2737 ["Upload MultiplEYE scanpath / fixation CSVs to begin."], 

2738 ) 

2739 

2740 words_raw, fix_raw = multipleye_frames_from_uploads( 

2741 fix_df, 

2742 aoi_df if not aoi_df.empty else None, 

2743 questions_df=questions_df if not questions_df.empty else None, 

2744 participant_meta_df=participant_df if not participant_df.empty else None, 

2745 ) 

2746 if fix_raw.empty: 

2747 problem = ( 

2748 "No MultiplEYE-shaped file names recognized — expected " 

2749 "`<session>_…_trial_N_<stimulus>_<scanpath|fixation>.csv` (e.g. " 

2750 "`001_ZH_CH_1_ET1_trial_1_Lit_Alchemist_4_scanpath.csv`). " 

2751 "Browser uploads keep the file name but drop the folder, so the " 

2752 "name must carry the session + stimulus." 

2753 ) 

2754 if active: 

2755 body.error(problem) 

2756 return _UploadResult( 

2757 empty_words_frame(), 

2758 empty_fixations_frame(), 

2759 pd.DataFrame(), 

2760 words_raw, 

2761 fix_raw, 

2762 [problem], 

2763 ) 

2764 

2765 has_words = not words_raw.empty 

2766 # Stimulus-level unless the recipe emitted per-reader boxes (question screens). 

2767 word_schema = multipleye_word_schema(words_raw) if has_words else None 

2768 fix_schema = dict( 

2769 MULTIPLEYE_FIX_SCHEMA, 

2770 word_id="word_idx" if "word_idx" in fix_raw.columns else None, 

2771 ) 

2772 # Carry MultiplEYE's trial-level facets + side-data through normalization 

2773 # (the upload path passes a keep-set, so registered meta fields must be named 

2774 # here to survive — the directory path uses keep=None and keeps them all). 

2775 keep_fix = compute_keep_columns( 

2776 fix_schema, 

2777 keep_columns={ 

2778 "genre", 

2779 "session", 

2780 "participant", 

2781 "is_practice", 

2782 "trial_num", 

2783 "comprehension_questions", 

2784 "pp_age", 

2785 "pp_gender", 

2786 "pp_native_language", 

2787 "pp_years_education", 

2788 "pp_education_level", 

2789 # DATA-24: a trial mixes reading pages (from the scanpath export) and 

2790 # question screens (from the fixation one), so the marker that tells 

2791 # them apart must survive the keep-set too. 

2792 "screen_kind", 

2793 }, 

2794 ) 

2795 keep_words = ( 

2796 compute_keep_columns( 

2797 word_schema, 

2798 keep_columns={ 

2799 "genre", 

2800 "comprehension_questions", 

2801 "screen_kind", 

2802 "aoi_block", 

2803 }, 

2804 ) 

2805 if has_words 

2806 else None 

2807 ) 

2808 # app._normalize_pair sets _composite_trial_columns from the (single-column) 

2809 # trial mapping → None, exactly what the reload branch expects. 

2810 try: 

2811 words_norm, fixations_norm = app._normalize_pair( 

2812 words_raw if has_words else empty_words_frame(), 

2813 word_schema, 

2814 fix_raw, 

2815 fix_schema, 

2816 keep_words=keep_words, 

2817 keep_fix=keep_fix, 

2818 ) 

2819 except Exception as exc: 

2820 # A rejected upload is a blocked wizard step, not a dead app — the 

2821 # uploaders above have to stay on screen for the user to fix the set of 

2822 # files they dropped in. See app.mapping_failure_problem. 

2823 problem = app.mapping_failure_problem(exc) 

2824 if active: 

2825 body.error(problem) 

2826 return _UploadResult( 

2827 empty_words_frame(), 

2828 empty_fixations_frame(), 

2829 pd.DataFrame(), 

2830 words_raw, 

2831 fix_raw, 

2832 [problem], 

2833 ) 

2834 filter_fields = ["genre", "session", "is_practice"] 

2835 st.session_state["wizard_filter_fields"] = filter_fields 

2836 schemas = {"words": word_schema, "fixations": fix_schema, "raw_gaze": None} 

2837 # DATA-66: with the loader-built frames' columns, so the map is stashed too. 

2838 app._stash_active_mapping( 

2839 "words", word_schema, words_raw.columns, keep_columns=keep_words 

2840 ) 

2841 app._stash_active_mapping( 

2842 "fixations", fix_schema, fix_raw.columns, keep_columns=keep_fix 

2843 ) 

2844 

2845 if active: 

2846 boxes_msg = ( 

2847 f" · **{len(words_norm):,}** word boxes" 

2848 if has_words 

2849 else " · no AOI boxes (upload *_aoi.csv for word boxes)" 

2850 ) 

2851 body.success( 

2852 f"~**{fixations_norm['participant_id'].nunique()}** participants · " 

2853 f"**{fixations_norm['trial_id'].nunique()}** page-trials" + boxes_msg 

2854 ) 

2855 st.session_state["_wizard_finalize_payload"] = { 

2856 "words": words_norm, 

2857 "fixations": fixations_norm, 

2858 "raw_gaze": pd.DataFrame(), 

2859 "filter_fields": filter_fields, 

2860 "composite_trial_columns": [], 

2861 "schemas": schemas, 

2862 # Share → Code: these schemas map the loader-built frames, not the 

2863 # files the user picked, so the snippet names the preset instead. 

2864 "source_recipe": {"schemas": {}, "steps": ["multipleye_preset"]}, 

2865 # DATA-66 — the loader-built frames' names (MultiplEYE's own files 

2866 # carry no identity columns; see the plan's settled defaults). 

2867 "column_names": for_tables( 

2868 schemas, 

2869 {"words": words_raw, "fixations": fix_raw}, 

2870 {"words": keep_words, "fixations": keep_fix}, 

2871 rewrites=st.session_state.get(app.HARMONIZE_REWRITES_KEY), 

2872 ), 

2873 "dropped_columns": { 

2874 "words": dropped_columns(words_raw, keep=keep_words) 

2875 if has_words 

2876 else [], 

2877 "fixations": dropped_columns(fix_raw, keep=keep_fix), 

2878 "raw_gaze": [], 

2879 }, 

2880 # CMP-8 §1: MultiplEYE needs no Recording-setup step — it *declares* 

2881 # its geometry. The corpus records a real 1920x1080 presentation 

2882 # monitor and stamps the stimulus font size + family from its own 

2883 # config onto the words, so both are `measured`. Only the physical 

2884 # size / viewing distance are unrecorded, and those stay honestly 

2885 # `assumed` rather than being invented as measured. 

2886 "setup": app.capture_setup_snapshot( 

2887 { 

2888 "screen": Provenance.MEASURED, 

2889 "geometry": Provenance.ASSUMED, 

2890 "text": Provenance.MEASURED, 

2891 } 

2892 ).to_dict(), 

2893 } 

2894 body.button( 

2895 f"{ICONS['confirm']} Add dataset", 

2896 type="primary", 

2897 key="wizard_finalize", 

2898 on_click=_finalize_wizard_dataset, 

2899 ) 

2900 

2901 return _UploadResult( 

2902 words_norm, fixations_norm, pd.DataFrame(), words_raw, fix_raw, [] 

2903 ) 

2904 

2905 

2906#: UX-55 r4 — `table name | Trial ID | Screen ID | Participant ID | Text ID | 

2907#: Word/IA ID | Fixation ID-or-Word text` — row 1 of the per-table block, now 

2908#: merged with what used to be a separate "geometry" section (r3/r4: 

2909#: identity-vs-description stopped paying for itself once Screen name left the 

2910#: view and Word/IA id joined the row it already read as identity). The name 

2911#: column stays narrow for one short word; the six pickers split the rest 

2912#: evenly — a column name is what has to stay readable, and six is the most 

2913#: this row fits. 

2914#: UX-127: the name column widened from 0.09 to 0.135 (and the CSS overlay's 

2915#: `width` in `styles.py` alongside it) — the file uploader's own "Browse 

2916#: files" button didn't fit inside the narrower column. The six picker cells 

2917#: shrink slightly (evenly) to make room. 

2918#: UX-129: widened again, from 0.135 to 0.155, *without* moving the CSS 

2919#: overlay's own `width` (still 13.5%, `styles.py`) — that mismatch is now 

2920#: deliberate. The overlay (and the border-right line on it) still ends at 

2921#: 13.5% of the block, but this reserved column is wider than that, so the 

2922#: extra ~2% sits empty between the line and the first picker cell, reading 

2923#: as breathing room rather than the pickers crowding the divider. 

2924_ID_ROW1_W = (0.155, 0.1409, 0.1409, 0.1409, 0.1409, 0.1409, 0.1409) 

2925 

2926#: Row 2 of the Fixations block: X · Y · Timestamp · Duration. Same grid as 

2927#: row 1 (UX-55 r2) so the two halves of the mapping line up down the page — 

2928#: four equal picker cells under the name column, since these selects hold 

2929#: column names rather than short ids. 

2930_FIX_ROW2_W = (0.155, 0.2113, 0.2113, 0.2113, 0.2113) 

2931 

2932#: Row 2 of the AOI block: the word box (a format radio plus four coordinate 

2933#: selects that lay themselves out) and, sharing the same line, Line index — 

2934#: the box gets most of the row, Line index the rest (UX-55 r3). 

2935_AOI_ROW2_W = (0.155, 0.678, 0.167) 

2936 

2937#: AN-32 — rows 3-4 of the AOI block: the reading measures the report brings, 

2938#: seven to a line under the same name column (thirteen fields on one line 

2939#: would leave each select a sliver). Shared with the ✏️ Edit dataset grid. 

2940MEASURE_ROW_W = (0.155, *([0.845 / 7] * 7)) 

2941#: The measures, split into those two lines: durations and the count first, 

2942#: then the flags, the regression count and the landing measures. 

2943MEASURE_ROWS = (READING_MEASURE_KEYS[:7], READING_MEASURE_KEYS[7:]) 

2944 

2945#: Row 2 of the Raw gaze block (UX-113): X · Y · Timestamp — no Duration, raw 

2946#: gaze has no such concept (unlike row 1, which reuses `_ID_ROW1_W` outright: 

2947#: same six identity fields, same shape as Fixations/AOI above it). 

2948_RAW_GAZE_ROW2_W = (0.155, 0.2817, 0.2817, 0.2816) 

2949 

2950#: UX-127: the metadata rows' own two-cell grid — same name-column width as 

2951#: every other table's row 1, one wide cell for the id-column + keep-fields 

2952#: picker stack (there is nothing to split across several picker cells here). 

2953_META_ROW_W = (0.155, 0.845) 

2954 

2955 

2956def _mark_add_attempted() -> None: 

2957 """Record that **✅ Add dataset** was pressed on a still-incomplete wizard. 

2958 

2959 UX-53's red state is deliberately not shown on arrival: a required field the 

2960 user has not reached yet is *unfilled*, not wrong. It turns red only once 

2961 they have asked for the dataset to be added. 

2962 """ 

2963 st.session_state[ADD_ATTEMPTED_KEY] = True 

2964 

2965 

2966def _wizard_name_header(host, active: bool) -> None: 

2967 """The dataset's name, its own numbered stage (UX-53 r8, UX-113). 

2968 

2969 It used to sit in step 7 beside *Add dataset*, i.e. below every mapping 

2970 field. The name is what the whole dataset is *called*, so it belongs to 

2971 neither the upload nor the mapping — it leads the wizard, in its own keyed 

2972 box that `styles.py` sizes up. 

2973 """ 

2974 if not active: 

2975 return 

2976 st.session_state.setdefault("wizard_dataset_name", _default_dataset_name()) 

2977 box = host.container(key="wiz_name_box") 

2978 box.text_input( 

2979 "Dataset name", 

2980 key="wizard_dataset_name", 

2981 help="Shown in the dataset list so you can switch back to it.", 

2982 placeholder="Name this dataset", 

2983 # Streamlit 1.65: a cleared name is not committed — the last one stays. 

2984 required=True, 

2985 # UX-113: the numbered stage heading above ("1 Dataset name") already 

2986 # says this — the widget's own label just repeated it verbatim. 

2987 label_visibility="collapsed", 

2988 ) 

2989 # UX-174 r2 — optional, and edited later on ✏️ Edit dataset. 

2990 box.text_area( 

2991 "Description", 

2992 key="wizard_dataset_description", 

2993 placeholder="Optional — what this dataset is: the participants, the texts, " 

2994 "the language.", 

2995 help=f"Shown under the dataset's name on the {ICONS['view_data']} Data Management page.", 

2996 height=68, 

2997 # Streamlit 1.65: read only when Add dataset runs, so no rerun per edit. 

2998 on_change="ignore", 

2999 ) 

3000 

3001 

3002def _wizard_statuses() -> dict[str, wizard_shell.StepStatus]: 

3003 """Each step's badge, derived from session state alone. 

3004 

3005 Deliberately cheap and frame-free: it runs at the *top* of the wizard, before 

3006 any step body renders, because the accordion seeding and every step's badge 

3007 need it up front. The one thing not in session state is this run's validation 

3008 result, so `_wizard_problems_last` carries the previous run's — the same 

3009 one-run-behind contract the old progress bar had. Any widget interaction 

3010 reruns the script, so a badge is never stale for longer than that, and the 

3011 **gate** on *Add dataset* is computed from live values at the end, never from 

3012 this. 

3013 """ 

3014 from scanpath_studio import metadata as metadata_mod 

3015 

3016 S = wizard_shell.StepStatus 

3017 ss = st.session_state 

3018 uploaded_core = bool(ss.get("col_map_fix_upload") or ss.get("col_map_words_upload")) 

3019 uploaded_any = uploaded_core or bool(ss.get("col_map_raw_gaze_upload")) 

3020 # Default to "pending" so the mapping step doesn't claim to be done before a 

3021 # single validation pass has run. 

3022 problems = ss.get("_wizard_problems_last", ["pending"]) 

3023 trial_mapped = bool( 

3024 ss.get("col_map_trial_unified") 

3025 or ss.get("col_map_fix_trial") 

3026 or ss.get("col_map_words_trial") 

3027 ) 

3028 setup_answered = all(ss.get(_SETUP_MODE_KEYS[g]) for g in SETUP_GROUPS) 

3029 # UX-114: "fields" no longer has widgets of its own — it reads whichever 

3030 # per-table keep picker(s) exist (`wizard_keep_<prefix>`, one per mapped 

3031 # table). 

3032 fields_touched = any( 

3033 ss.get(f"wizard_keep_{prefix}") 

3034 for prefix in ("col_map_words", "col_map_fix", "col_map_raw_gaze") 

3035 ) 

3036 

3037 def required(done: bool) -> wizard_shell.StepStatus: 

3038 if done: 

3039 return S.DONE 

3040 # ACTION ("blocked on something specific") only once there is data to act 

3041 # on; before that the step is simply not started. 

3042 return S.ACTION if uploaded_any else S.TODO 

3043 

3044 statuses = { 

3045 "name": S.DONE if (ss.get("wizard_dataset_name") or "").strip() else S.TODO, 

3046 "identity": required(trial_mapped), 

3047 "geometry": required(uploaded_any and not problems), 

3048 "setup": required(setup_answered), 

3049 # DATA-20: optional until a table is actually attached. Reads the parsed 

3050 # frame, not the uploader widget — the frame is what survives a rerun. 

3051 "readers": ( 

3052 S.DONE if ss.get(metadata_mod.RAW_SESSION_KEY) is not None else S.OPTIONAL 

3053 ), 

3054 # UX-114: "fields" is not a numbered part any more (its pickers moved 

3055 # into "mapping", one per table) — the key stays for any bookkeeping 

3056 # that still badges it by name, and it's OPTIONAL so it never blocks 

3057 # "data" from reaching DONE below. 

3058 "fields": S.DONE if fields_touched else S.OPTIONAL, 

3059 } 

3060 # UX-53/UX-113/UX-114/UX-129: "identity"/"geometry" are *sections* nested 

3061 # inside the "data" step (folded from the old "mapping" step, which used 

3062 # to be a numbered stage of its own), so it aggregates them — a part is 

3063 # only DONE when every required section under it is, and only once 

3064 # something has been uploaded at all. The section keys stay in the dict 

3065 # because the section headings, the blocker list and 

3066 # `_wizard_problems_last` all still badge per topic. 

3067 statuses["data"] = ( 

3068 S.DONE 

3069 if uploaded_any and all(statuses[k] is S.DONE for k in ("identity", "geometry")) 

3070 else required(False) 

3071 ) 

3072 return statuses 

3073 

3074 

3075def _render_data_setup(active: bool) -> _UploadResult: 

3076 """The Add-dataset wizard: seven steps in an accordion (DATA-22). 

3077 

3078 Replaces a single 620-line function that rendered 16 expanders across five 

3079 numbered subsections under a three-step progress bar matching neither. The 

3080 steps are now the unit of everything — the badge, the progress chip, the 

3081 guide's target, and the "go here to fix it" button on a blocker. 

3082 

3083 ``active`` is the guided wizard; ``active=False`` is the compact collapsed 

3084 *Data & mapping* review panel, where `wizard_shell.step_panel` degrades each 

3085 step to a plain heading (that panel is itself an expander, and Streamlit 

3086 forbids nesting one inside another). 

3087 

3088 **Every step body renders on every run, open or closed.** Streamlit drops a 

3089 widget's key at the end of any run in which the widget did not render, and 

3090 `controls.column_mapping_ui` builds its `col_map_*` widgets without 

3091 `persist_state` — so gating a body on its expander being open would silently 

3092 discard that step's mapping. Collapsed-but-rendered is the contract. 

3093 """ 

3094 statuses = _wizard_statuses() 

3095 if active: 

3096 # UX-66 — ONE row, and it stays put while the page scrolls: the title, 

3097 # the guide button, the docs link, and the way out. Everything else that 

3098 # used to stack above the wizard is gone (the page header and its summary 

3099 # in `app.main`, the "Adding a dataset…" caption in 

3100 # `resolve_data_source`, and the second copy of this very title 

3101 # that lived here) — four headings for one screen, from three modules. 

3102 # 

3103 # Keyed so `styles.py` can pin it; the CSS is scoped to this key alone, 

3104 # so only the add-dataset screen gets a sticky bar. 

3105 bar = st.container(key="wiz_sticky_bar") 

3106 # Help and Cancel sit together at the right, each as wide as its 

3107 # label: the help pill used to stretch over a column half again as wide 

3108 # as Cancel's, a long empty pill beside the way out. 

3109 title_col, actions_col = bar.columns([7.2, 3.6], vertical_alignment="center") 

3110 help_col = cancel_col = actions_col.container( 

3111 horizontal=True, 

3112 horizontal_alignment="right", 

3113 vertical_alignment="center", 

3114 gap="small", 

3115 ) 

3116 title_col.markdown( 

3117 '<div class="sps-wiz-title">Set up your dataset</div>', 

3118 unsafe_allow_html=True, 

3119 ) 

3120 # Step-by-step guide: a bottom-right card that auto-opens once per session 

3121 # and is replayable via the popover below. Arm it (auto/first-visit) then 

3122 # render the card early so it streams before the heavy upload/normalize 

3123 # work. 

3124 maybe_show_wizard_guide() 

3125 render_spotlight_wizard_guide() 

3126 # UX-84: one ❓ Help popover replaces the two buttons that used to sit 

3127 # here (🧭 guide · 📖 docs) — a popover, not a dialog, since it is a 

3128 # two-item chooser with no modal weight to it (matches #UX-65's nav 

3129 # Help, minus the "arm-then-bounce" dance that menu entries need). 

3130 # "Setup help", not "Help": the nav's ❓ Help is on screen too, and this 

3131 # one holds only the setup guide and the loading-data docs. 

3132 # Its two entries are menu rows — tertiary, icon + label, left-aligned 

3133 # — and the same kind of control as each other: a content-width 

3134 # button stacked over a stretched link button read as two unrelated 

3135 # widgets of two different widths. 

3136 with help_col.popover(f"{ICONS['help']} Setup help", width="content"): 

3137 render_wizard_guide_button(st) 

3138 # A real `link_button`, not an in-app navigation: it opens in a new 

3139 # tab and so cannot lose an in-progress upload the way switching 

3140 # views would — the same reason #BUG-31's leave prompt exists. 

3141 st.link_button( 

3142 # UX-66 r2: named for what it *is* rather than for the page it 

3143 # opens — "Data guide" reads like one more wizard step on a row 

3144 # of wizard controls, which is the one thing it is not. 

3145 "More documentation ↗", 

3146 "https://lacclab.github.io/scanpath-studio/guides/loading-data/", 

3147 icon=ICONS["docs"], 

3148 type="tertiary", 

3149 help="What your export needs, how this wizard maps it, and the " 

3150 "recording setup it asks for.", 

3151 ) 

3152 # The way out, on the row that stays on screen. 

3153 # 

3154 # BUG-36: checked by return value, not `on_click` — arms the same 

3155 # leave-prompt the nav-triggered leave uses (`_render_leave_prompt` 

3156 # below reads it later in this same run), rather than calling 

3157 # `app.leave_add_data_wizard` straight away, which discarded an 

3158 # in-progress upload with no confirmation at all. It has to be the 

3159 # return-value form here too: `app.main`'s hold-the-view prologue pops 

3160 # `WIZARD_LEAVE_KEY` at the top of every run whenever the nav is 

3161 # already on 🗂️ Data — true for the whole time the wizard is open — so 

3162 # an `on_click` callback (which runs *before* that prologue) would have 

3163 # its flag wiped before `_render_leave_prompt` ever saw it. Setting it 

3164 # here, after the prologue has already run this pass, is what makes it 

3165 # stick for the `_render_leave_prompt(bar)` call three lines down. 

3166 if cancel_col.button( 

3167 "✕ Cancel", 

3168 key="cancel_add_data", 

3169 # #374 F30: secondary. UX-66 r2 made it the same filled blue as 

3170 # ✅ Add dataset, which put the page's loudest button on the way 

3171 # out, at the top, before anything had been added. 

3172 type="secondary", 

3173 help="Leave the wizard and go back to the dataset you were on.", 

3174 width="content", 

3175 ): 

3176 st.session_state[WIZARD_LEAVE_KEY] = _VIEW_DATA 

3177 _render_leave_prompt(bar) 

3178 # UX-53 r8 / UX-113: no progress chips. They were navigation for an 

3179 # accordion that no longer exists — the five parts are linear, so there 

3180 # is nothing to flip between and a chip row would be a menu with one 

3181 # path through it. 

3182 body = st.container() 

3183 else: 

3184 panel = st.expander(f"{ICONS['data_mapping']} Data & mapping", expanded=False) 

3185 body = panel 

3186 if panel.button( 

3187 f"{ICONS['settings']} Change dataset / mapping", 

3188 key="wizard_reconfigure", 

3189 help="Re-open the setup wizard.", 

3190 ): 

3191 st.session_state["setup_complete"] = False 

3192 st.rerun() 

3193 

3194 # Five linear parts. `part()` while the wizard is active — a one-line 

3195 # numbered headline, no expander — and `step_panel`'s bold-heading 

3196 # degradation in the collapsed *Data & mapping* review panel, which is 

3197 # itself an expander and so can nest neither. 

3198 def _part(step_id: str, *, trailing=None): 

3199 step = wizard_shell.STEPS_BY_ID[step_id] 

3200 status = statuses.get(step_id, wizard_shell.StepStatus.TODO) 

3201 if active: 

3202 # UX-88: no status badge. `step_panel` already refuses one (a keyed 

3203 # expander that changes label remounts collapsed); the headline was 

3204 # the last place a ⚠️ could nag about a field visible on the same 

3205 # screen. 

3206 return wizard_shell.part(body, step, trailing=trailing) 

3207 return wizard_shell.step_panel(body, step, status, active=False) 

3208 

3209 # UX-53 r8 / UX-113: the dataset's **name** is its own numbered stage — it 

3210 # names the whole thing, so it belongs to neither the upload nor the 

3211 # mapping — rendered first since nothing after it makes sense without one. 

3212 s_name = _part("name") 

3213 _wizard_name_header(s_name, active) 

3214 

3215 def _render_restore_trigger(host) -> None: 

3216 # UX-127: beside stage 2's title now, not stage 3's — UX-113's reason 

3217 # (it never touches the uploads themselves) no longer separates the 

3218 # two stages, since every table now uploads *inside* stage 3 too; 

3219 # what actually matters is that a restored setup is visible before 

3220 # the wizard is filled in, and stage 2 is the first thing on screen. 

3221 restore_box = host.popover(f"{ICONS['undo']} Restore a saved setup (optional)") 

3222 _wizard_restore_config(restore_box) 

3223 _render_restored_config_caption(restore_box) 

3224 

3225 # UX-114: the "Dataset format" choice + the MultiplEYE branch it dispatches 

3226 # to are held back this release (mirrors PRE-21/PRE-22's gate) — the code 

3227 # stays for a later revival, but with the flag off there is no format 

3228 # *question* at all, on either Add or Edit dataset (this function backs 

3229 # both). Force the key rather than relying on the `.get(..., "Generic")` 

3230 # fallback below, so a stale "MultiplEYE" left over from an earlier, 

3231 # flagged session can't sneak back in. 

3232 _multipleye_enabled = multipleye_upload_enabled() 

3233 if not _multipleye_enabled: 

3234 st.session_state["wizard_dataset_format"] = "Generic" 

3235 

3236 # Restoring a config seeds `col_map_*` keys, which mean nothing on the 

3237 # MultiplEYE branch (no column mapping there) — read from state, same as 

3238 # the format-dispatch check below, since the picker itself hasn't 

3239 # (re-)rendered yet at this point in the script. 

3240 _generic_format = ( 

3241 st.session_state.get("wizard_dataset_format", "Generic") != "MultiplEYE" 

3242 ) 

3243 s1 = _part( 

3244 "data", 

3245 trailing=_render_restore_trigger if active and _generic_format else None, 

3246 ) 

3247 # UX-129: "mapping" is no longer its own numbered stage — everything that 

3248 # used to render under it (the identity/geometry sections, each table's 

3249 # own upload+mapping row) now renders straight into `s1`, the same "data" 

3250 # part. Kept as its own name (`s_map`, not just reusing `s1` everywhere 

3251 # below) because the two halves' content is still worth telling apart by 

3252 # name in the code that follows, even though they share one heading now. 

3253 s_map = s1 

3254 

3255 # === Upload data tables: intro note ====================================== 

3256 if active and _multipleye_enabled: 

3257 s1.segmented_control( 

3258 "Dataset format", 

3259 ["Generic", "MultiplEYE"], 

3260 key="wizard_dataset_format", 

3261 default="Generic", 

3262 help="**Generic**: map your own columns. **MultiplEYE**: upload the " 

3263 "corpus's scanpath/fixation + AOI CSVs and the app parses identity " 

3264 "from the file names (no column mapping needed).", 

3265 ) 

3266 

3267 # A dedicated dataset format runs its own tailored flow and bypasses the 

3268 # generic mapping steps. It renders into `body`, NOT into a step panel: 

3269 # `_render_multipleye_upload` opens its own canvas expander, and 

3270 # expander-in-expander is forbidden. Read from state so the collapsed review 

3271 # panel branches the same way. 

3272 if ( 

3273 _multipleye_enabled 

3274 and st.session_state.get("wizard_dataset_format", "Generic") == "MultiplEYE" 

3275 ): 

3276 return _render_multipleye_upload(body, active) 

3277 

3278 # The run-locally tip keeps its always-created container: a *conditional* 

3279 # child here shifts the element tree mid-parse and Streamlit leaves a 

3280 # greyed-out ghost of the whole upload group on screen (BUG-2). 

3281 intro = s1.container() 

3282 core_uploaded = bool( 

3283 st.session_state.get("col_map_fix_upload") 

3284 or st.session_state.get("col_map_words_upload") 

3285 ) 

3286 already_uploaded = core_uploaded or bool( 

3287 st.session_state.get("col_map_raw_gaze_upload") 

3288 ) 

3289 if active and not already_uploaded: 

3290 # UX-113: a small caption near the stage title, not a boxed alert — the 

3291 # three tables below say the same thing at their own titles' hover, this 

3292 # is just the nudge to open with. 

3293 guide = intro.container( 

3294 key="wiz_example_row", 

3295 horizontal=True, 

3296 vertical_alignment="center", 

3297 gap="small", 

3298 ) 

3299 guide.caption( 

3300 f"{ICONS['upload']} Upload at least one of **Fixations**, " 

3301 f"**{WORDS_TABLE_LABEL}**, or " 

3302 "**Raw gaze** below to get started.", 

3303 width="content", 

3304 ) 

3305 # DATA-67: a tiny AOI + fixation pair that maps with no manual pick, 

3306 # with a README naming every column's unit and what the IDs mean. 

3307 # Built on click (`data=` a callable) and `on_click="ignore"`, so the 

3308 # download neither costs a run nor reruns the wizard. 

3309 guide.download_button( 

3310 "Download example tables", 

3311 data=example_import_zip, 

3312 file_name=EXAMPLE_ZIP_FILE, 

3313 mime="application/zip", 

3314 icon=ICONS["download"], 

3315 key="wizard_example_download", 

3316 on_click="ignore", 

3317 help="Two tiny tables, one Words table and one fixation table, that " 

3318 "import with every column mapped automatically. The README inside " 

3319 "explains each column, its unit, and the IDs.", 

3320 ) 

3321 app_url = str(getattr(st.context, "url", "") or "") 

3322 if not is_loopback_url(app_url): 

3323 intro.markdown( 

3324 f"{ICONS['tip']} **Working with your own data?** Use the " 

3325 f"[desktop app]({CITATION['desktop_url']}) ↗ (Windows, macOS, " 

3326 "Linux): it keeps your data on your computer, is faster, and " 

3327 "handles much larger datasets than this hosted copy. Or install " 

3328 "it with pip:\n\n" 

3329 "```bash\npip install scanpath-studio\nscanpath-studio\n```" 

3330 ) 

3331 

3332 # UX-124: the size/type line `st.file_uploader` prints under its own 

3333 # dropzone ("5GB per file • CSV, TSV, …") doesn't fit this narrow column 

3334 # either — `styles.py` hides it there, so it needs to survive somewhere: 

3335 # appended to `help_text`, which already reaches both the title's hover 

3336 # tooltip and the uploader's own accessible help. 

3337 _upload_types_note = ( 

3338 ", ".join(t.upper() for t in app._UPLOAD_TYPES) 

3339 + f" — up to {upload_limit_label()} per file." 

3340 ) 

3341 

3342 def upload_box( 

3343 host, 

3344 *, 

3345 label, 

3346 help_text, 

3347 prefix, 

3348 multi, 

3349 noun, 

3350 kind=None, 

3351 short_label=None, 

3352 notes_host=None, 

3353 ): 

3354 # UX-113: the title reads like every mapping field's — dotted underline, 

3355 # the description on hover (`.sps-fhelp`) — instead of Streamlit's own 

3356 # label + native (~1s) help tooltip. UX-123: `short_label` is what 

3357 # actually shows — the narrow row-name column this now renders into 

3358 # (UX-122) is too tight for "Fixations table(s)"/"Words / IA 

3359 # table(s)", which just ellipsised. The accessible name and the 

3360 # uploader's own tooltip still carry the full `label`/`help_text`. 

3361 inline_field_label(host, short_label or label, help_text, emphasis=True) 

3362 frame = app._read_uploaded_frame( 

3363 uploader_label=label, 

3364 upload_help=help_text, 

3365 state_prefix=prefix, 

3366 multi=multi, 

3367 container=host, 

3368 kind=kind, 

3369 label_visibility="collapsed", 

3370 ) 

3371 if not frame.empty: 

3372 # PERF-6 parses only the columns the mapping needs, so the frame's 

3373 # own width is the *plan*, not the file's. Count the header, which 

3374 # is what the user is being asked to check against their export. 

3375 n_columns = len(app._uploaded_header(prefix)) or len(frame.columns) 

3376 # UX-117/118/119: a small caption, like the metadata tables' own 

3377 # "✓ N identified" — the row/column count was a full-width green 

3378 # banner, disproportionate next to a one-line join caption. The 

3379 # preview trigger and the count sit in one `stats` container 

3380 # (`styles.py` turns it into a packed flex row, the same trick 

3381 # `railbtn_*` uses) rather than `st.columns` — a ratio-based 

3382 # column always reserves its ratio's share of the row even once 

3383 # its content is `width="content"`-sized, which is what left a gap 

3384 # between an icon-sized button and the text that used to follow 

3385 # it two columns later. 

3386 stats = host.container(key=f"wiz_upload_stats_{prefix}") 

3387 if active: 

3388 # UX-117: the preview used to sit permanently on the page — 

3389 # several rows tall per table, three tables wide. A hover-only 

3390 # reveal was tried first, but `st.dataframe`'s canvas grid 

3391 # (glide-data-grid) sizes itself once at mount via 

3392 # ResizeObserver and never recovers from mounting inside a 

3393 # zero-size/hidden box, so it stayed blank even once "shown". 

3394 # A popover sidesteps that: Streamlit doesn't render its body 

3395 # at all until opened, so the grid always mounts visible. 

3396 # UX-119: icon-only trigger (no "Preview" label) — way smaller, 

3397 # matching the rail's other icon-only popovers (⇅, ✏️). 

3398 # UX-200: named for screen readers; `styles.py` clips it. 

3399 preview = stats.popover( 

3400 "Preview the first rows", 

3401 icon=ICONS["preview"], 

3402 width="content", 

3403 wrap=True, 

3404 help="Preview — first rows", 

3405 key=f"iconpop_preview_{prefix}", 

3406 ) 

3407 preview.caption("First rows:") 

3408 preview.dataframe(frame.head(), width="stretch", hide_index=True) 

3409 # UX-124: two short lines, not one that has to wrap mid-count in 

3410 # this narrow column — and no leading "✓", which read as a stray 

3411 # mark once split from a sentence it no longer shares a line with. 

3412 counts = stats.container(key=f"wiz_upload_counts_{prefix}") 

3413 counts.caption(plural(len(frame), noun)) 

3414 counts.caption(plural(n_columns, "column")) 

3415 # #374 F3: a zip holding both EyeLink reports reads only the ones 

3416 # this row takes — say which, and where the rest go. 

3417 # Drawn beside the row (`notes_host`), not in this narrow column. 

3418 for note in app.upload_zip_notes( 

3419 st.session_state.get(f"{prefix}_upload"), kind 

3420 ): 

3421 (notes_host or host).caption(f"{ICONS['info']} {note}") 

3422 return frame 

3423 

3424 # UX-122/UX-127/UX-129: none of the six tables upload at the top of this 

3425 # stage — each has its own uploader further down, in its own mapping row 

3426 # (see `row_fix`/`row_words`/`rg_row1`/`_render_metadata_uploads` below), 

3427 # replacing that table's row-name label. `raw_fix`/`raw_words`/`raw_gaze` 

3428 # are placeholders until those rows render; the top of this stage holds 

3429 # only the intro note. 

3430 raw_fix, raw_words, raw_gaze = pd.DataFrame(), pd.DataFrame(), pd.DataFrame() 

3431 

3432 def _render_metadata_uploads(word_schema, fix_schema) -> None: 

3433 """DATA-20/DATA-29's participant + trial tables, plus the text table — 

3434 UX-127: three more rows of the same "left = upload, right = mapping" 

3435 format Fixations/AOI/Raw gaze use above, under a small **Metadata** 

3436 heading, rather than the three-wide block of their own this used to 

3437 be in stage 2. `word_schema`/`fix_schema` are `{}` pre-upload — there 

3438 is nothing to derive a join report from yet, which is also the 

3439 correct answer — and the real mapping once identity (stage 3) has 

3440 resolved it. Still only while `active`: the collapsed *Data & 

3441 mapping* review panel would otherwise build the same widget keys as 

3442 the 🗂️ Data page's section. 

3443 """ 

3444 if not active: 

3445 return 

3446 from scanpath_studio.tabs import ( 

3447 render_participant_metadata_section, 

3448 render_text_metadata_section, 

3449 render_trial_metadata_section, 

3450 ) 

3451 

3452 meta_host.markdown( 

3453 '<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True 

3454 ) 

3455 # UX-129: its own keyed container, not just `meta_host` directly — 

3456 # `emphasis=True` alone made this read as a fourth table name beside 

3457 # Fixations/AOI/Raw gaze above it, rather than the section heading 

3458 # for the three rows below it. `styles.py` gives this specific title 

3459 # its own, more prominent look instead of adding a third 

3460 # `inline_field_label` style shared with (and so muddying) every 

3461 # other caller. Confined to the left column (`_META_ROW_W[0]`, same 

3462 # split `_row_body` uses) and centered within it, rather than 

3463 # spanning — and centering across — the whole row: this heading 

3464 # belongs to the Participants/Trials/Texts *titles* below it, which 

3465 # live in that same narrow column, not to the wide mapping side. 

3466 meta_heading_row = meta_host.columns( 

3467 [_META_ROW_W[0], 1 - _META_ROW_W[0]], gap="small" 

3468 ) 

3469 meta_heading = meta_heading_row[0].container(key="wiz_map_meta_heading") 

3470 inline_field_label( 

3471 meta_heading, 

3472 "Metadata", 

3473 "Optional per-participant, per-trial and per-text tables. Once " 

3474 "attached, their columns behave like fields in the data: " 

3475 "filters, chips, trial sorting, inspection and export.", 

3476 emphasis=True, 

3477 ) 

3478 

3479 def _meta_row(slug, renderer, ids): 

3480 block = meta_host.container(key=f"wiz_map_block_meta_{slug}") 

3481 row = block.columns(_META_ROW_W, gap="small") 

3482 # UX-116: `live_join=False` — there is no finished dataset to join 

3483 # against yet (the pools below are provisional, still shifting as 

3484 # identity mapping is worked out), so the wizard only collects the 

3485 # upload + id-column + keep-fields choices here. The real join 

3486 # runs once, in `_finalize_wizard_dataset`, against the dataset's 

3487 # own settled pools — see `tabs.commit_deferred_metadata`. 

3488 renderer( 

3489 ids, 

3490 host=row[1], 

3491 live_join=False, 

3492 upload_host=row[0].container(key=f"wiz_map_upload_meta_{slug}"), 

3493 ) 

3494 

3495 _meta_row( 

3496 "participant", 

3497 render_participant_metadata_section, 

3498 _wizard_reader_ids(raw_words, word_schema, raw_fix, fix_schema), 

3499 ) 

3500 meta_host.markdown( 

3501 '<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True 

3502 ) 

3503 _meta_row( 

3504 "trial", 

3505 render_trial_metadata_section, 

3506 _wizard_trial_combos(raw_words, word_schema, raw_fix, fix_schema), 

3507 ) 

3508 meta_host.markdown( 

3509 '<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True 

3510 ) 

3511 _meta_row( 

3512 "text", 

3513 render_text_metadata_section, 

3514 _wizard_text_ids(raw_words, word_schema, raw_fix, fix_schema), 

3515 ) 

3516 

3517 # === Upload data tables: per-table upload + mapping rows ================ 

3518 # UX-129: folded into this same stage — no longer a separate numbered 

3519 # "Map data fields" heading. 

3520 

3521 # UX-71: this screen's dropdowns hold column names, so their option lists 

3522 # get to be wider than the selects they drop from (see the docstring — it is 

3523 # per-screen because the menu is portalled out of our DOM and cannot be 

3524 # scoped any other way). 

3525 s_map.markdown(mapping_menu_css(), unsafe_allow_html=True) 

3526 

3527 # Reserve-then-fill, in reading order: the mapping sections, then the counts, 

3528 # then the footer. UX-67 moved the trial/reader/text counts down here from 

3529 # under the identity rows -- they are a check you run *before committing*, 

3530 # so they belong beside "Add dataset", not three sections above it. 

3531 sections_host = s_map.container() 

3532 counts_host = s_map.container() 

3533 

3534 # UX-113: no heading — it is the first, usually only thing under "3 Map 

3535 # data fields" now that `setup`/`fields` are their own stages (only a 

3536 # raw-gaze upload adds a second, titled "Raw gaze" section below), so a 

3537 # sub-heading here just repeated the stage title above it. 

3538 s2 = sections_host.container() 

3539 

3540 # UX-122: each table's own uploader replaces its plain row-name label — 

3541 # there is no separate "Upload data files" step for these three tables 

3542 # any more (only the metadata tables still upload in stage 2), so the row 

3543 # — and its uploader — has to exist before we know whether that table has 

3544 # data. Fixations then AOI always render, in that fixed order, each 

3545 # resolving `has_fix`/`has_words` and reserving its own feature/keep rows 

3546 # immediately (UX-89: a table's rows stay adjacent, not batched by kind) 

3547 # before the next table's row begins. 

3548 id_rows = {} 

3549 feature_rows = {} 

3550 extra_rows = {} 

3551 keep_rows = {} 

3552 

3553 # UX-123: "Derive columns from the filename" stays the first thing in 

3554 # this stage, as it was before UX-122 — reserved here (screen order is 

3555 # creation order) and filled in further down, once the uploads below it 

3556 # have actually run and there is something to derive from. 

3557 derive_host = s2.container() 

3558 derive_gap = s2.container() 

3559 

3560 # UX-125: each table's own block (row 1 + row 2 + keep-picker) is wrapped 

3561 # in its own container so the uploader — `position: absolute` inside it, 

3562 # see `styles.py` — can center against the *whole* block's height, not 

3563 # just row 1's (which the uploader itself was already taller than, 

3564 # defeating `vertical_alignment="center"` on row 1 alone). 

3565 fix_block = s2.container(key="wiz_map_block_col_map_fix") 

3566 # #374 F12: which export goes in this row, in EyeLink's own terms. 

3567 fix_note = _row_note(fix_block, ROW_CAPTIONS["fixations"]) 

3568 row_fix = fix_block.columns(_ID_ROW1_W, gap="small", vertical_alignment="center") 

3569 raw_fix = upload_box( 

3570 row_fix[0].container(key="wiz_map_upload_col_map_fix"), 

3571 label="Fixations table(s)", 

3572 short_label="Fixations", 

3573 help_text="One row per fixation (EyeLink Fixation Report); files stack. " 

3574 + _upload_types_note, 

3575 prefix="col_map_fix", 

3576 multi=True, 

3577 noun="fixation", 

3578 kind="fixations", 

3579 notes_host=fix_note, 

3580 ) 

3581 has_fix = not raw_fix.empty 

3582 if has_fix: 

3583 id_rows["fix"] = row_fix[1:] 

3584 feature_rows["fix"] = fix_block.columns( 

3585 _FIX_ROW2_W, gap="small", vertical_alignment="bottom" 

3586 ) 

3587 keep_rows["fix"] = fix_block.container() 

3588 

3589 s2.markdown('<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True) 

3590 words_block = s2.container(key="wiz_map_block_col_map_words") 

3591 words_note = _row_note(words_block, ROW_CAPTIONS["words"]) 

3592 row_words = words_block.columns( 

3593 _ID_ROW1_W, gap="small", vertical_alignment="center" 

3594 ) 

3595 raw_words = upload_box( 

3596 row_words[0].container(key="wiz_map_upload_col_map_words"), 

3597 label=f"{WORDS_TABLE_LABEL} table(s)", 

3598 short_label=WORDS_TABLE_LABEL, 

3599 help_text="One row per word box (EyeLink Interest Area Report); files " 

3600 "stack. " + _upload_types_note, 

3601 prefix="col_map_words", 

3602 kind="words", 

3603 multi=True, 

3604 noun="word", 

3605 notes_host=words_note, 

3606 ) 

3607 has_words = not raw_words.empty 

3608 if has_words: 

3609 id_rows["words"] = row_words[1:] 

3610 feature_rows["words"] = words_block.columns( 

3611 _AOI_ROW2_W, gap="small", vertical_alignment="bottom" 

3612 ) 

3613 # AN-32: the two measure lines, reserved here so they sit under the 

3614 # box row and above the character-AOI toggle, whatever fills first. 

3615 measure_rows = [ 

3616 words_block.columns(MEASURE_ROW_W, gap="small", vertical_alignment="bottom") 

3617 for _ in MEASURE_ROWS 

3618 ] 

3619 extra_rows["words"] = words_block.container() 

3620 keep_rows["words"] = words_block.container() 

3621 

3622 # The raw-gaze block's upload, here rather than further down where its 

3623 # pickers are drawn: *Derive columns from the filename* runs before those 

3624 # pickers and has to see this table too, or its Table picker could never 

3625 # offer Raw gaze. `s3` is still created after `s2`, so it still draws below 

3626 # the Fixations and AOI blocks. 

3627 # UX-104 — the raw-gaze block. UX-113: same "name column + evenly split 

3628 # pickers" grid as the Fixations/AOI blocks above (a single generic 

3629 # `column_mapping_ui` grid read as a cramped, differently-shaped block 

3630 # beside them). UX-122: its own uploader replaces the "Raw gaze" label 

3631 # in row 1's name column, so — like Fixations/AOI above — row 1 always 

3632 # renders (there is nowhere else to upload); row 2 and everything below 

3633 # only once there is something to map. 

3634 # UX-125: keyed like `fix_block`/`words_block` above — the raw-gaze 

3635 # uploader centers against this whole block's height too. 

3636 s3 = sections_host.container(key="wiz_map_block_col_map_raw_gaze") 

3637 # UX-127: reserved here, right after `s3` (raw gaze) — a sibling of `s2`/ 

3638 # `s3` in `sections_host`, so whatever `_render_metadata_uploads` fills 

3639 # into it later lands after all three main tables' rows in the DOM, 

3640 # regardless of how late in the script it actually runs. 

3641 meta_host = sections_host.container() 

3642 # UX-129: unconditional now — UX-125/127's `min-height` fix means the 

3643 # Fixations/AOI blocks above always render a visible block even with 

3644 # nothing uploaded, so raw gaze is never actually "the first block" any 

3645 # more (the `has_words or has_fix` guard this used to carry was stale; 

3646 # without it, AOI and Raw gaze had no line between them when both were 

3647 # still empty). 

3648 s3.markdown('<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True) 

3649 # Row 1: Trial ID · Screen ID · Participant ID · Text ID · Word/IA ID · 

3650 # Word text/label — same six-cell grid, same field order, as the 

3651 # Fixations/AOI row above. 

3652 rg_row1 = s3.columns(_ID_ROW1_W, gap="small", vertical_alignment="center") 

3653 raw_gaze = upload_box( 

3654 rg_row1[0].container(key="wiz_map_upload_col_map_raw_gaze"), 

3655 label="Raw gaze table (optional)", 

3656 short_label="Raw gaze", 

3657 # VIZ-45: raw gaze can be the dataset's only table, not just an 

3658 # overlay — and nothing is derived from it, which is worth saying 

3659 # before someone uploads samples expecting fixations back. 

3660 help_text="Sample-level gaze (one file), drawn as recorded — under the " 

3661 "fixations, or on its own as the dataset's only table. No fixations " 

3662 "are detected from it. " + _upload_types_note, 

3663 prefix="col_map_raw_gaze", 

3664 multi=False, 

3665 noun="gaze point", 

3666 ) 

3667 

3668 # UX-113: stages 3-5 render unconditionally now, rather than exiting here 

3669 # before any of them exist — every `has_words`/`has_fix`/`raw_gaze.empty` 

3670 # guard below already tolerates all three being empty (the same guards the 

3671 # raw-gaze-only path needed), so there is nothing left to special-case. 

3672 # `raw_gaze` itself uploads a little further down (its own row), so 

3673 # `nothing_uploaded` is finalized there. 

3674 prop_w = ( 

3675 _c_propose_word_schema(raw_words, frame_fingerprint(raw_words)) 

3676 if has_words 

3677 else {} 

3678 ) 

3679 prop_f = ( 

3680 _c_propose_fix_schema(raw_fix, frame_fingerprint(raw_fix)) if has_fix else {} 

3681 ) 

3682 word_schema: dict = {} 

3683 fix_schema: dict = {} 

3684 

3685 # Column derivation must run *before* the identifier pickers so the 

3686 # derived columns are mappable below. UX-53 took it out of its popover — 

3687 # "advanced" is a reason to place a control last, not to hide it behind a 

3688 # click — so it renders inline, below the pickers it feeds. UX-113: no 

3689 # longer gated on `source_file` specifically — the tool derives from any 

3690 # uploaded column now, so any non-empty table is reason enough to offer 

3691 # it. UX-122/123: it needs the Fixations/AOI uploads above to already 

3692 # exist, which by this point in the script they do — but it still 

3693 # *renders* above them, into `derive_host`/`derive_gap`, reserved before 

3694 # either row so screen order puts it first regardless of fill order. 

3695 # What the files themselves held — a mapped column outside it was made from 

3696 # the file names, which Share → Code has to name (its loader has no such 

3697 # step). Read before the derive step below adds its columns. 

3698 uploaded_columns = { 

3699 "words": set(map(str, raw_words.columns)), 

3700 "fixations": set(map(str, raw_fix.columns)), 

3701 "raw_gaze": set(map(str, raw_gaze.columns)), 

3702 } 

3703 if has_words or has_fix or not raw_gaze.empty: 

3704 # UX-129: the same nudge the top of the stage shows before anything 

3705 # is uploaded, repeated here above "Derive columns from the 

3706 # filename" — once one table is in, this is the next thing on 

3707 # screen, and a reader who uploaded only one of the three tables 

3708 # still benefits from the reminder that each is optional on its own 

3709 # but at least one is required. 

3710 derive_host.caption( 

3711 f"{ICONS['upload']} Each table is optional; the dataset needs at least one." 

3712 ) 

3713 raw_words, raw_fix, raw_gaze = _wizard_filename_derive( 

3714 derive_host, 

3715 raw_words, 

3716 raw_fix, 

3717 raw_gaze, 

3718 ) 

3719 derive_gap.markdown( 

3720 '<div class="sps-wiz-blockgap"></div>', unsafe_allow_html=True 

3721 ) 

3722 

3723 if has_words or has_fix: 

3724 # `_render_identity_field` takes its cells in (fixations, AOI) order. 

3725 def _cells_for(index: int) -> list: 

3726 return [id_rows[s][index] for s in ("fix", "words") if s in id_rows] 

3727 

3728 id_extras = counts_host 

3729 # Row 1, in the order the request pins: Trial ID · Screen ID · 

3730 # Participant ID · Text ID · Word/IA ID · Fixation ID-or-Word text. 

3731 disjoint_trials = _wizard_trial_step( 

3732 s2, 

3733 raw_words, 

3734 raw_fix, 

3735 prop_w, 

3736 prop_f, 

3737 word_schema, 

3738 fix_schema, 

3739 has_words, 

3740 has_fix, 

3741 cells=_cells_for(0), 

3742 ) 

3743 # Screen ID (DATA-21 multipart) — a simple per-table field, not a 

3744 # composite like Trial/Participant/Text, so it goes straight through 

3745 # `_map_section` rather than `_render_identity_field`. Screen *name* 

3746 # (`screen_index`) is dropped from the view entirely (UX-55 r4): order 

3747 # within a multipart trial still comes from it when it is mapped and 

3748 # from first appearance when it is not (multipart.normalize_screen_ 

3749 # identity), so the column stays auto-detected — see Background for 

3750 # why the field itself is gone rather than merely unlabelled. 

3751 screen_specs = ( 

3752 ("fix", raw_fix, FIX_FIELD_SPECS, prop_f, fix_schema, has_fix), 

3753 ("words", raw_words, WORD_FIELD_SPECS, prop_w, word_schema, has_words), 

3754 ) 

3755 for slug, raw, specs, proposal, schema, present in screen_specs: 

3756 if not present or slug not in id_rows: 

3757 continue 

3758 schema.update( 

3759 _map_section( 

3760 raw, 

3761 specs, 

3762 proposal, 

3763 f"col_map_{slug}", 

3764 id_rows[slug][1], 

3765 ["screen_id"], 

3766 ) 

3767 ) 

3768 _wizard_participant_text_step( 

3769 "participant", 

3770 "Participant ID", 

3771 "participant", 

3772 "The participant column — or several to build one. Leave empty for one " 

3773 "participant (Fixations) or boxes shared by all (Words).", 

3774 s2, 

3775 raw_words, 

3776 raw_fix, 

3777 prop_w, 

3778 prop_f, 

3779 word_schema, 

3780 fix_schema, 

3781 has_words, 

3782 has_fix, 

3783 cells=_cells_for(2), 

3784 extras_host=id_extras, 

3785 ) 

3786 _wizard_participant_text_step( 

3787 "text_id", 

3788 "Text ID", 

3789 "text", 

3790 "The text column — or several to compose an id. Map your item " 

3791 "column here if trial order was randomized: `TRIAL_INDEX` only " 

3792 "orders a participant's trials. A Words table with no Participant " 

3793 "ID attaches to a trial by it when the Trial IDs differ. Leave " 

3794 "empty to use the Trial ID (a repeated trial keeps the same text).", 

3795 s2, 

3796 raw_words, 

3797 raw_fix, 

3798 prop_w, 

3799 prop_f, 

3800 word_schema, 

3801 fix_schema, 

3802 has_words, 

3803 has_fix, 

3804 cells=_cells_for(3), 

3805 extras_host=id_extras, 

3806 ) 

3807 # DATA-49: an AOI table with no Participant ID is stimulus-level and 

3808 # attaches by Text ID when the trial ids differ, so disjoint trial ids 

3809 # are no mapping error there: the join is stated (or refused) above 

3810 # Add dataset instead. 

3811 if disjoint_trials and word_schema.get("participant"): 

3812 id_extras.warning(disjoint_trials) 

3813 # Word/IA ID — the fixations table's own `word_id` says which AOI a 

3814 # fixation hit; the AOI table's is which AOI a row *is*. Different 

3815 # columns, same slot: both tables read it as identity now. 

3816 for slug, raw, specs, proposal, schema, present in screen_specs: 

3817 if not present or slug not in id_rows: 

3818 continue 

3819 schema.update( 

3820 _map_section( 

3821 raw, 

3822 specs, 

3823 proposal, 

3824 f"col_map_{slug}", 

3825 id_rows[slug][4], 

3826 ["word_id"], 

3827 ) 

3828 ) 

3829 # Row 1's last slot differs per table: Fixation ID for Fixations, 

3830 # Word text/label for AOI. 

3831 if has_fix: 

3832 fix_schema.update( 

3833 _map_section( 

3834 raw_fix, 

3835 FIX_FIELD_SPECS, 

3836 prop_f, 

3837 "col_map_fix", 

3838 id_rows["fix"][5], 

3839 ["fixation_id"], 

3840 ) 

3841 ) 

3842 if has_words: 

3843 word_schema.update( 

3844 _map_section( 

3845 raw_words, 

3846 WORD_FIELD_SPECS, 

3847 prop_w, 

3848 "col_map_words", 

3849 id_rows["words"][5], 

3850 ["text"], 

3851 ) 

3852 ) 

3853 

3854 # Row 2 of each block: the table's own features, filled into the cells 

3855 # reserved above so they sit directly under that table's identity row. 

3856 # UX-89 also removed the per-block validation warnings that used to 

3857 # print here ("Words/IA — missing Word/IA ID", …): a required field that 

3858 # is empty turns red in place the moment ✅ Add dataset is pressed, and 

3859 # a sentence repeating it below the row was the third copy of the same 

3860 # complaint on a page whose problem is length. 

3861 if has_fix: 

3862 for cell, key in zip( 

3863 feature_rows["fix"][1:], ["x", "y", "timestamp", "duration"] 

3864 ): 

3865 fix_schema.update( 

3866 _map_section( 

3867 raw_fix, FIX_FIELD_SPECS, prop_f, "col_map_fix", cell, [key] 

3868 ) 

3869 ) 

3870 if has_words: 

3871 # The box (a format radio plus four coordinate selects that lay 

3872 # themselves out) and Line index share the row (UX-55 r3). 

3873 words_row2 = feature_rows["words"] 

3874 word_schema.update( 

3875 _map_section( 

3876 raw_words, 

3877 WORD_FIELD_SPECS, 

3878 prop_w, 

3879 "col_map_words", 

3880 words_row2[1], 

3881 ["box"], 

3882 ) 

3883 ) 

3884 word_schema.update( 

3885 _map_section( 

3886 raw_words, 

3887 WORD_FIELD_SPECS, 

3888 prop_w, 

3889 "col_map_words", 

3890 words_row2[2], 

3891 ["line"], 

3892 ) 

3893 ) 

3894 # AN-32 — the reading measures, two lines named once. Each is an 

3895 # optional field seeded from its EyeLink name, so an IA report maps 

3896 # them all without a click and a report without them leaves the 

3897 # lines empty rather than hiding them behind a switch. 

3898 # No name in the row's first column: on this screen it holds the 

3899 # AOI uploader, centred on the whole block, and a label there 

3900 # printed over the file card. Each field names its measure. 

3901 for row, keys in zip(measure_rows, MEASURE_ROWS): 

3902 for cell, key in zip(row[1:], keys): 

3903 word_schema.update( 

3904 _map_section( 

3905 raw_words, 

3906 WORD_FIELD_SPECS, 

3907 prop_w, 

3908 "col_map_words", 

3909 cell, 

3910 [key], 

3911 ) 

3912 ) 

3913 # UX-104 — line 3 of the AOI block. One row per *character* is a 

3914 # fact about this table, so the question sits with the fields that 

3915 # describe it, not in a later section the user reads after they 

3916 # have stopped thinking about the AOI file. UX-118: indented to 

3917 # where the pickers above start (`_row_body`), not the row-name 

3918 # label — this line has no name of its own, it describes the AOI 

3919 # table above it. 

3920 aoi_extra = _row_body(extra_rows["words"]) 

3921 aggregate_char_boxes_on = aoi_extra.toggle( 

3922 "Merge character boxes into word boxes", 

3923 key="wizard_aggregate_char_boxes", 

3924 help="For a Words table with one row per *character* (e.g. Chinese " 

3925 "or Japanese): merge the characters sharing a Trial ID and " 

3926 "Word/IA ID into one word box.", 

3927 ) 

3928 # UX-113: only relevant once aggregating — a table whose rows are 

3929 # grouped into sub-screen blocks that each restart their own word 

3930 # numbering (e.g. a comprehension question's answer blocks) needs 

3931 # to say so, or two blocks' word 0 would silently merge into one 

3932 # box (see data.aggregate_char_boxes). 

3933 if aggregate_char_boxes_on: 

3934 word_schema.update( 

3935 _map_section( 

3936 raw_words, 

3937 WORD_FIELD_SPECS, 

3938 prop_w, 

3939 "col_map_words", 

3940 aoi_extra, 

3941 ["block"], 

3942 ) 

3943 ) 

3944 

3945 # UX-113: stages 3-5 render unconditionally now, rather than exiting here 

3946 # before any of them exist — every `has_words`/`has_fix`/`raw_gaze.empty` 

3947 # guard below already tolerates all three being empty (the same guards the 

3948 # raw-gaze-only path needed), so there is nothing left to special-case. 

3949 nothing_uploaded = raw_words.empty and raw_fix.empty and raw_gaze.empty 

3950 prop_g = ( 

3951 _c_propose_raw_gaze_schema(raw_gaze, frame_fingerprint(raw_gaze)) 

3952 if not raw_gaze.empty 

3953 else {} 

3954 ) 

3955 if not raw_gaze.empty: 

3956 # Row 2: X · Y · Timestamp — no Duration, raw gaze has no such concept. 

3957 rg_row2 = s3.columns(_RAW_GAZE_ROW2_W, gap="small", vertical_alignment="bottom") 

3958 raw_gaze_schema: dict = {} 

3959 row1_keys = ["trial", "screen_id", "participant", "text_id", "word_id", "text"] 

3960 for cell, key in zip(rg_row1[1:], row1_keys): 

3961 raw_gaze_schema.update( 

3962 _map_section( 

3963 raw_gaze, 

3964 RAW_GAZE_FIELD_SPECS, 

3965 prop_g, 

3966 "col_map_raw_gaze", 

3967 cell, 

3968 [key], 

3969 ) 

3970 ) 

3971 for cell, key in zip(rg_row2[1:], ["x", "y", "timestamp"]): 

3972 raw_gaze_schema.update( 

3973 _map_section( 

3974 raw_gaze, 

3975 RAW_GAZE_FIELD_SPECS, 

3976 prop_g, 

3977 "col_map_raw_gaze", 

3978 cell, 

3979 [key], 

3980 ) 

3981 ) 

3982 # UX-120: the same per-table "extra fields to keep" picker 

3983 # Fixations/AOI have — raw gaze had none, because 

3984 # `normalize_raw_gaze` had no mechanism to carry an extra column 

3985 # through at all (it built an entirely fresh frame from only the 

3986 # schema-mapped columns). It does now (`_carry_extra_columns`), so 

3987 # this table gets the same choice, no registry of known-optional 

3988 # fields to offer (there is no raw-gaze equivalent of 

3989 # `saccade_amplitude` common enough to earn a canonical name) — 

3990 # every unmapped column is offered as a plain extra to keep. 

3991 rg_kept, rg_meta = _wizard_table_keep_picker( 

3992 s3.container(), 

3993 raw_gaze, 

3994 raw_gaze_schema, 

3995 [], 

3996 "col_map_raw_gaze", 

3997 noun="Raw gaze", 

3998 ) 

3999 else: 

4000 raw_gaze_schema = {} 

4001 rg_kept, rg_meta = set(), [] 

4002 

4003 words_problems = validate_word_schema(word_schema) if has_words else [] 

4004 fix_problems = validate_fix_schema(fix_schema) if has_fix else [] 

4005 raw_gaze_problems = ( 

4006 validate_raw_gaze_schema(raw_gaze_schema) if not raw_gaze.empty else [] 

4007 ) 

4008 problems: list = [] 

4009 if words_problems: 

4010 line = "Words table: " + "; ".join(words_problems) + "." 

4011 if any(p.startswith("need either") for p in words_problems): 

4012 # #374 F12: say what to do, not only what is missing. 

4013 line += ( 

4014 " Without word boxes the text can't be drawn: map the box " 

4015 "columns, re-export the Interest Area Report with IA_LEFT, " 

4016 "IA_RIGHT, IA_TOP and IA_BOTTOM, or remove this table to see " 

4017 "fixations alone." 

4018 ) 

4019 problems.append(line) 

4020 if fix_problems: 

4021 problems.append("Fixations: " + "; ".join(fix_problems)) 

4022 # A raw-gaze-ONLY upload: an incomplete raw-gaze mapping is the only thing 

4023 # blocking a usable dataset — fold it into `problems` so finalize is gated. 

4024 if not has_words and not has_fix and raw_gaze_problems: 

4025 problems.append("Raw gaze: " + "; ".join(raw_gaze_problems)) 

4026 st.session_state["_wizard_problems_last"] = list(problems) 

4027 

4028 # UX-53 r5 removed the auto-detect summary card ("✓ Trial id — Trial_Id", 

4029 # "⚠️ Word box — not detected", …). It restated, in a block of its own, what 

4030 # every field now shows in place: the ✨ flag and the select's tint say 

4031 # detected-or-not per row, and a genuinely missing required field turns red 

4032 # and is listed above ✅ Add dataset. Since the fields all live on one screen 

4033 # now, the summary was a second copy of what was directly below it. 

4034 

4035 # UX-114: "Keep extra fields" is no longer a stage of its own — each 

4036 # table's own decision now sits directly under that table's own mapping 

4037 # (the `keep_rows[slug]` containers reserved above), so there is nothing 

4038 # left to render at this point except collect the two tables' results. 

4039 keep_by_prefix: dict = {} 

4040 filter_fields: list = [] 

4041 if has_words: 

4042 kept, meta = _wizard_table_keep_picker( 

4043 keep_rows["words"], 

4044 raw_words, 

4045 word_schema, 

4046 WORD_OPTIONAL_FIELDS, 

4047 "col_map_words", 

4048 noun=WORDS_TABLE_LABEL, 

4049 ) 

4050 keep_by_prefix["col_map_words"] = kept 

4051 filter_fields += meta 

4052 if has_fix: 

4053 kept, meta = _wizard_table_keep_picker( 

4054 keep_rows["fix"], 

4055 raw_fix, 

4056 fix_schema, 

4057 FIX_OPTIONAL_FIELDS, 

4058 "col_map_fix", 

4059 noun="Fixations", 

4060 ) 

4061 keep_by_prefix["col_map_fix"] = kept 

4062 filter_fields += meta 

4063 keep_by_prefix["col_map_raw_gaze"] = rg_kept 

4064 filter_fields += rg_meta 

4065 # The same dest field (e.g. "difficulty") can legitimately come from both 

4066 # tables — `_wizard_table_keep_picker` decides per table, so de-dup here, 

4067 # order-preserving. 

4068 st.session_state["wizard_filter_fields"] = list(dict.fromkeys(filter_fields)) 

4069 

4070 # DATA-20's participant table (and DATA-29's trial/text tables) render as 

4071 # three more rows of this same stage (UX-127) — they are uploads, and 

4072 # they belong with the others. Called down here, after identity (stage 3) 

4073 # has resolved the real mapping, so the join report reads the up-to-date 

4074 # schema rather than the `{}` an earlier call would have had to use. 

4075 _render_metadata_uploads(word_schema, fix_schema) 

4076 

4077 # UX-113: "Recording setup" is stage 5 — its own numbered part. Its old 

4078 # caption ("These describe the screen the data was recorded on…") now lives 

4079 # as the WizardStep's own hover caption. 

4080 s_setup = _part("setup") 

4081 restored_setup = _restored_setup_snapshot() 

4082 if restored_setup is not None: 

4083 _apply_restored_setup(restored_setup) 

4084 s_setup.caption( 

4085 "✓ Pre-answered from the restored setup file — review it below." 

4086 ) 

4087 

4088 # DATA-46: the estimate needs canonical coordinates, and nothing is 

4089 # normalized yet — project the mapped geometry columns rather than hand over 

4090 # the raw upload, whose `IA_LEFT` / `CURRENT_FIX_X` it cannot read. 

4091 def _estimate() -> tuple[int, int]: 

4092 words = raw_words if has_words else None 

4093 fixations = raw_fix if has_fix else None 

4094 return _c_estimate_canvas( 

4095 words, 

4096 word_schema if has_words else None, 

4097 fixations, 

4098 fix_schema if has_fix else None, 

4099 (frame_fingerprint(words), frame_fingerprint(fixations)), 

4100 ( 

4101 _geometry_key( 

4102 word_schema if has_words else None, _WORD_GEOMETRY_FIELDS 

4103 ), 

4104 _geometry_key(fix_schema if has_fix else None, _FIX_GEOMETRY_FIELDS), 

4105 ), 

4106 ) 

4107 

4108 setup_snapshot = _wizard_setup_step( 

4109 s_setup, raw_words, raw_fix, has_boxes=has_words, estimate=_estimate 

4110 ) 

4111 

4112 # The foot of the wizard: what is still missing, then the button. UX-53 put 

4113 # the alerts *directly above* **Add dataset** — a blocker listed a screen 

4114 # away from the control it blocks is a blocker the user reads after 

4115 # clicking. UX-113/UX-129: trails every stage, not just the upload/mapping 

4116 # one. 

4117 s6 = body.container() 

4118 

4119 setup_blockers = [ 

4120 _SETUP_HEADINGS[g] for g, p in setup_snapshot.provenance.items() if p is None 

4121 ] 

4122 # `nothing_uploaded` is its own term (not folded into `problems`/ 

4123 # `setup_blockers`): a restored setup config can answer every group with 

4124 # nothing uploaded, which must still block — there is nothing to add. 

4125 blocked = nothing_uploaded or bool(problems) or bool(setup_blockers) 

4126 

4127 # UX-53 dropped the review table. Every figure in it is now stated where it 

4128 # is decided — row counts beside each upload, the trial count under the trial 

4129 # picker (`_wizard_trial_step`), the mapped column beside its own field, and 

4130 # each setup value beside its provenance radio. Repeating them here made a 

4131 # second screen out of things the user had just read. 

4132 

4133 # UX-88: no "Still to do" list, and no status badges on the part headlines 

4134 # or section headings above. The page said the same thing three times — a 

4135 # badge on the part, a badge on its section, and a warning down here — for a 

4136 # field the user can see is empty, on a page whose entire complaint has been 

4137 # length. What is left is the one thing that actually points at the problem: 

4138 # clicking ✅ Add dataset sets `ADD_ATTEMPTED_KEY`, which turns every unmapped 

4139 # required row **red in place**. `blocked` still gates finalizing; it just 

4140 # no longer narrates. 

4141 

4142 if problems: 

4143 if active: 

4144 # UX-88 removed the *Still to do* list that used to print here on 

4145 # arrival. What it must NOT remove is the answer to "I pressed Add 

4146 # and nothing happened" — and for some blockers there is nothing 

4147 # else to see: a raw-gaze-only upload whose trial id cannot be 

4148 # mapped has no required field on screen to turn red, so with no 

4149 # message at all the button is a dead end. 

4150 # 

4151 # So the problems still get stated, on exactly the terms UX-90 set 

4152 # for the Recording-setup gate: red, and only once the user has 

4153 # actually tried. Before that the page stays quiet. 

4154 if st.session_state.get(ADD_ATTEMPTED_KEY): 

4155 for line in problems: 

4156 s6.error(line) 

4157 # Enabled, not disabled (UX-53). A disabled button cannot be *tried*, 

4158 # and "red when you try to add with it empty" needs the attempt: the 

4159 # click sets ADD_ATTEMPTED_KEY, which is what turns every unmapped 

4160 # required row red on the rerun. It still cannot finalize — this 

4161 # branch returns the problems either way. 

4162 _wizard_footer( 

4163 s6, 

4164 disabled=False, 

4165 on_click=_mark_add_attempted, 

4166 help_text="Some required fields are still empty — click to mark " 

4167 "them in red.", 

4168 ) 

4169 st.session_state["_composite_trial_columns"] = None 

4170 return _UploadResult( 

4171 empty_words_frame(), 

4172 empty_fixations_frame(), 

4173 pd.DataFrame(), 

4174 raw_words, 

4175 raw_fix, 

4176 problems, 

4177 ) 

4178 

4179 # Record the mapping so the Data Inspection tab shows it once the wizard is 

4180 # collapsed (active=False) and the tabs render with this upload. 

4181 wizard_schemas = { 

4182 "words": dict(word_schema) if has_words else None, 

4183 "fixations": dict(fix_schema) if has_fix else None, 

4184 "raw_gaze": dict(raw_gaze_schema) if not raw_gaze.empty else None, 

4185 } 

4186 for table, schema in wizard_schemas.items(): 

4187 app._stash_active_mapping(table, schema) 

4188 

4189 # Char→word aggregation: collapse character-level AOIs to one box per word 

4190 # using the final word mapping, before normalization (which expects one row 

4191 # per word box). 

4192 if has_words and st.session_state.get("wizard_aggregate_char_boxes"): 

4193 characters = raw_words 

4194 raw_words = _c_aggregate_char_boxes( 

4195 characters, 

4196 word_schema, 

4197 frame_fingerprint(characters), 

4198 _schema_key(word_schema), 

4199 ) 

4200 # BUG-103: a fresh copy out of the cache each rerun, named by its input. 

4201 assign_derived( 

4202 raw_words, "aggregate_char_boxes", characters, _schema_key(word_schema) 

4203 ) 

4204 

4205 keep_words = ( 

4206 compute_keep_columns( 

4207 word_schema, keep_columns=keep_by_prefix.get("col_map_words", set()) 

4208 ) 

4209 if has_words 

4210 else None 

4211 ) 

4212 keep_fix = ( 

4213 compute_keep_columns( 

4214 fix_schema, keep_columns=keep_by_prefix.get("col_map_fix", set()) 

4215 ) 

4216 if has_fix 

4217 else None 

4218 ) 

4219 if has_words or has_fix: 

4220 try: 

4221 words_norm, fixations_norm = app._normalize_pair( 

4222 raw_words, 

4223 word_schema if has_words else None, 

4224 raw_fix, 

4225 fix_schema if has_fix else None, 

4226 keep_words=keep_words, 

4227 keep_fix=keep_fix, 

4228 ) 

4229 except Exception as exc: 

4230 # The mapping is complete but the pipeline rejects the combination 

4231 # (see app.mapping_failure_problem). Blocked exactly like an 

4232 # incomplete one: the wizard stays up, ✅ Add dataset stays off, and 

4233 # the review step's badge reads from `_wizard_problems_last`. 

4234 problem = app.mapping_failure_problem(exc) 

4235 st.session_state["_wizard_problems_last"] = [problem] 

4236 st.session_state["_composite_trial_columns"] = None 

4237 if active: 

4238 s6.error(problem) 

4239 _wizard_footer(s6, disabled=True, help_text=problem) 

4240 return _UploadResult( 

4241 empty_words_frame(), 

4242 empty_fixations_frame(), 

4243 pd.DataFrame(), 

4244 raw_words, 

4245 raw_fix, 

4246 [problem], 

4247 ) 

4248 else: 

4249 # Raw-gaze-only dataset — record composite-trial columns from the raw-gaze 

4250 # mapping so the trial picker still offers one selector per component. 

4251 words_norm, fixations_norm = empty_words_frame(), empty_fixations_frame() 

4252 rg_trial_cols = ( 

4253 trial_mapping_columns(raw_gaze_schema["trial"]) 

4254 if raw_gaze_schema and raw_gaze_schema.get("trial") 

4255 else [] 

4256 ) 

4257 st.session_state["_composite_trial_columns"] = ( 

4258 rg_trial_cols if len(rg_trial_cols) > 1 else None 

4259 ) 

4260 

4261 if active: 

4262 # BUG-54 / BUG-56: a complete mapping can still meet rows it cannot 

4263 # use — a numeric column that did not parse (a decimal-comma export, a 

4264 # text column picked as a coordinate), a row with no trial id. The load 

4265 # carries on without them, so say which and what was done, where the 

4266 # other blockers are: directly above ✅ Add dataset. 

4267 tables = ( 

4268 ("Words table", raw_words, word_schema, has_words), 

4269 ("Fixations", raw_fix, fix_schema, has_fix), 

4270 ) 

4271 for table, raw, schema, present in tables: 

4272 if not present: 

4273 continue 

4274 for line in _c_normalization_issues( 

4275 raw, schema, frame_fingerprint(raw), _schema_key(schema), table 

4276 ): 

4277 s6.warning(f"{ICONS['warning']} {line}") 

4278 

4279 # DATA-49: an AOI table with no Participant ID is shared by the readings 

4280 # of its texts, so say which key attached it — the trial ID or the Text 

4281 # ID — and how many readings got boxes. A join that reaches none never 

4282 # gets here: `_normalize_pair` raised, and the error above blocks the add. 

4283 join = ( 

4284 st.session_state.get(app.STIMULUS_JOIN_KEY) if has_words and has_fix else None 

4285 ) 

4286 if active and join is not None: 

4287 # Anything worth acting on — readings without boxes, Text IDs that 

4288 # disagree with their boxes' — is a warning, never the green caption. 

4289 if join.needs_warning: 

4290 s6.warning(f"{ICONS['warning']} {join.describe()}") 

4291 else: 

4292 s6.caption(f"{ICONS['confirm']} {join.describe()}") 

4293 

4294 if ( 

4295 active 

4296 and has_words 

4297 and has_fix 

4298 and _readers_do_not_line_up(words_norm, fixations_norm) 

4299 ): 

4300 # BUG-59: the trial-id check above compares trial ids alone, so a pair 

4301 # of tables that share every trial but spell the readers differently 

4302 # passed it — and every scanpath then drew over no text. 

4303 s6.warning( 

4304 f"{ICONS['warning']} The two tables share Trial IDs but no participant: no fixation's " 

4305 "participant + trial has word boxes, so every scanpath would be " 

4306 "drawn without its text. Check that **Participant ID** names the " 

4307 "same participants, spelled the same way, in both tables." 

4308 ) 

4309 

4310 raw_gaze_norm = pd.DataFrame() 

4311 if not raw_gaze.empty: 

4312 if raw_gaze_problems: 

4313 s3.warning("Raw gaze ignored — " + "; ".join(raw_gaze_problems)) 

4314 else: 

4315 raw_gaze_norm = normalize_raw_gaze( 

4316 raw_gaze, 

4317 raw_gaze_schema, 

4318 keep_columns=keep_by_prefix.get("col_map_raw_gaze", set()), 

4319 ) 

4320 

4321 if active: 

4322 # Stash the assembled, already-normalized dataset so the finalize callback 

4323 # can store it. The callback (not an inline `if button:` handler) is what 

4324 # makes "Add dataset" reliable: a real st.file_uploader in the wizard can 

4325 # swallow an inline button click (the click reruns, the uploader 

4326 # re-renders, and the handler is never reached), so the dataset would 

4327 # never get stored. on_click runs as part of the click event, before the 

4328 # rerun — exactly like the "➕ Add data" button. 

4329 st.session_state["_wizard_finalize_payload"] = { 

4330 "words": words_norm, 

4331 "fixations": fixations_norm, 

4332 "raw_gaze": raw_gaze_norm, 

4333 "filter_fields": list(st.session_state.get("wizard_filter_fields", [])), 

4334 # Persist the composite trial-id components (session-only state, not in 

4335 # the frames) so switching back restores the cascading picker. 

4336 "composite_trial_columns": list( 

4337 st.session_state.get("_composite_trial_columns") or [] 

4338 ), 

4339 # Persist the column mapping so reselecting this stored dataset can 

4340 # repopulate the Data Inspection tab's mapping table. 

4341 "schemas": wizard_schemas, 

4342 # Share → Code: how a script loads these files — the mapping as 

4343 # chosen here, in the files' own names, and what this screen did 

4344 # that `load_scanpath_data` cannot replay. 

4345 "source_recipe": _source_recipe( 

4346 wizard_schemas, 

4347 uploaded_columns, 

4348 aggregated=has_words 

4349 and bool(st.session_state.get("wizard_aggregate_char_boxes")), 

4350 ), 

4351 # Source columns discarded at normalization — surfaced as a note in 

4352 # the Data Inspection remap editor (they can't be remapped without a 

4353 # re-upload). set(raw.columns) - keep is exactly the dropped set. 

4354 "dropped_columns": { 

4355 "words": dropped_columns(raw_words, keep=keep_words) 

4356 if has_words 

4357 else [], 

4358 "fixations": dropped_columns(raw_fix, keep=keep_fix) if has_fix else [], 

4359 "raw_gaze": dropped_columns(raw_gaze, schema=raw_gaze_schema) 

4360 if not raw_gaze.empty 

4361 else [], 

4362 }, 

4363 # CMP-8 §1 / DATA-22 §7: the geometry this dataset was set up with, 

4364 # plus how each group came to be known. A stored upload recorded no 

4365 # geometry at all before this, which is why switching to one left the 

4366 # canvas on the previous source's monitor. 

4367 "setup": setup_snapshot.to_dict(), 

4368 # DATA-66: what each canonical column was called in these files — 

4369 # the record the app shows, exports and accepts names from. Built 

4370 # from exactly what normalization read: the tables after character 

4371 # aggregation, narrowed to the kept columns. 

4372 "column_names": for_tables( 

4373 wizard_schemas, 

4374 { 

4375 "words": raw_words, 

4376 "fixations": raw_fix, 

4377 # Raw gaze the wizard ignored (a broken mapping) is not 

4378 # stored, so it gets no names either. 

4379 "raw_gaze": raw_gaze if not raw_gaze_norm.empty else None, 

4380 }, 

4381 {"words": keep_words, "fixations": keep_fix}, 

4382 # And the columns whose values the load changed. 

4383 rewrites=st.session_state.get(app.HARMONIZE_REWRITES_KEY), 

4384 ), 

4385 } 

4386 for table, names in st.session_state["_wizard_finalize_payload"][ 

4387 "column_names" 

4388 ].items(): 

4389 app._stash_active_mapping( 

4390 table, 

4391 wizard_schemas.get(table), 

4392 names=ColumnNames.from_payload(names), 

4393 ) 

4394 # UX-53: the two things you can do with a finished setup share one row — 

4395 # save it for next time, or add it — instead of stacking two full-width 

4396 # buttons. UX-93 made that row the same on all three endings. 

4397 _wizard_footer( 

4398 s6, 

4399 disabled=blocked, 

4400 on_click=_finalize_wizard_dataset, 

4401 help_text=( 

4402 f"Upload a Fixations, {WORDS_TABLE_LABEL} or Raw gaze table above " 

4403 "to get started." 

4404 if nothing_uploaded 

4405 else "Answer Recording setup first: " + ", ".join(setup_blockers) 

4406 if setup_blockers 

4407 else "Store this dataset and switch to it." 

4408 ), 

4409 ) 

4410 

4411 return _UploadResult( 

4412 words_norm, fixations_norm, raw_gaze_norm, raw_words, raw_fix, [] 

4413 ) 

4414 

4415 

4416# ----------------------------------------------------------------------------- 

4417# Main application 

4418# -----------------------------------------------------------------------------