Coverage for scanpath_studio/annotations.py: 94%

514 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Per-trial researcher annotations: favorites (stars), tags, and free notes. 

2 

3Annotations are keyed by ``(participant_id, trial_id)`` and live in Streamlit 

4session state, so they persist across reruns within a session. There is no 

5backend; a local run keeps them in the recovery cache, and to share them 🗂️ 

6Data → **Annotations** exports and imports a dataset's as JSON (UX-174). The 

7Scanpath view's Export bundle can include the same file (UX-179). 

8 

9**DATA-48 — annotations belong to a dataset**, as the metadata tables do since 

10DATA-47: :data:`ANNOTATIONS_STATE_KEY` holds the *selected* dataset's store only 

11(so every reader — the per-trial editor, the pickers' markers, the ⭐ / tag 

12filters, the Data page, the Export bundle — reads it unchanged), and every other 

13dataset's waits in :data:`DATASET_STORE_KEY`. :func:`activate_dataset`, called 

14by ``app.main`` beside ``metadata.activate_dataset``, swaps them when the 

15selection changes. Two datasets that reuse ``(participant, trial)`` ids 

16therefore no longer share a star, a tag or a note. 

17 

18The module is split into a *pure* core (``records_to_store`` / 

19``store_to_records`` / ``serialize`` / ``deserialize`` — no Streamlit, unit 

20tested) and a thin session-backed layer plus the small render helpers used by 

21``tabs.py`` and ``app.py``. 

22""" 

23 

24from __future__ import annotations 

25 

26import json 

27from collections.abc import Iterable 

28 

29import pandas as pd 

30import streamlit as st 

31 

32from .constants import ICONS, upload_limit_mb 

33from .data import respell_reading 

34from .fields import PANEL_LABEL_W, panel_field, row_label 

35from .session_keys import COMPARE_SOURCE_STATE_KEY 

36 

37ANNOTATIONS_STATE_KEY = "trial_annotations" 

38#: The annotations file (🗂️ Data → Annotations, an Export bundle's 

39#: ``annotations.json``). Schema 3 (DATA-48) adds the optional ``dataset`` the 

40#: file was exported from — for the reader, not the importer: a file of any 

41#: schema imports into the dataset that is open, the only one it can belong to. 

42SCHEMA_VERSION = 3 

43 

44# Per-trial annotation widgets use this prefix so they can be cleared on import 

45# (forcing a re-seed from the freshly loaded store), and on a dataset swap. 

46_WIDGET_PREFIX = "annotrial_" 

47 

48# Always-available tag suggestions (users can add their own on top). 

49PRESET_TAGS = ["To exclude", "Review", "Good example", "Check alignment"] 

50 

51# Parent entries use ``(participant_id, trial_id)``; screen entries add a third 

52# ``screen_id`` component. Keeping parent keys unchanged preserves filtering and 

53# every schema-1 annotation sidecar. 

54Key = tuple[str, ...] 

55Entry = dict[str, object] 

56 

57 

58# --------------------------------------------------------------------------- 

59# Pure core (no Streamlit) — unit tested in tests/test_annotations.py 

60# --------------------------------------------------------------------------- 

61 

62 

63def default_entry() -> Entry: 

64 return {"star": False, "tags": [], "note": ""} 

65 

66 

67def _normalize_entry(star: object, tags: object, note: object) -> Entry: 

68 # A recovery-cache record is not checked like an imported file is 

69 # (`deserialize`), so a stray scalar here is no tags rather than a crash. 

70 tags = tags if isinstance(tags, (list, tuple, set, frozenset)) else [] 

71 clean_tags = sorted({str(t).strip() for t in tags if str(t).strip()}) 

72 return {"star": bool(star), "tags": clean_tags, "note": str(note or "").strip()} 

73 

74 

75def is_empty_entry(entry: Entry) -> bool: 

76 """True when an entry carries no information (and can be dropped).""" 

77 return ( 

78 not entry.get("star") 

79 and not entry.get("tags") 

80 and not str(entry.get("note") or "").strip() 

81 ) 

82 

83 

84def records_to_store(records: list[dict]) -> dict[Key, Entry]: 

85 """Build a ``{(pid, tid): entry}`` store from a list of flat records.""" 

86 store: dict[Key, Entry] = {} 

87 for rec in records or []: 

88 pid = rec.get("participant_id") 

89 tid = rec.get("trial_id") 

90 if pid is None or tid is None: 

91 continue 

92 entry = _normalize_entry( 

93 rec.get("star", False), rec.get("tags", []), rec.get("note", "") 

94 ) 

95 if not is_empty_entry(entry): 

96 screen_id = rec.get("screen_id") 

97 key = ( 

98 (str(pid), str(tid), str(screen_id)) 

99 if screen_id not in (None, "") 

100 else (str(pid), str(tid)) 

101 ) 

102 store[key] = entry 

103 return store 

104 

105 

106def store_to_records(store: dict[Key, Entry]) -> list[dict]: 

107 """Flatten a store into a sorted list of records for JSON export.""" 

108 records = [] 

109 for key, entry in sorted(store.items()): 

110 pid, tid, *screen = key 

111 record = { 

112 "participant_id": pid, 

113 "trial_id": tid, 

114 "star": bool(entry.get("star", False)), 

115 "tags": list(entry.get("tags", [])), 

116 "note": str(entry.get("note", "")), 

117 } 

118 if screen: 

119 record["screen_id"] = screen[0] 

120 records.append(record) 

121 return records 

122 

123 

124def serialize(store: dict[Key, Entry], *, dataset: str | None = None) -> str: 

125 """Serialize a store to a JSON document string. 

126 

127 ``dataset`` names the dataset the annotations were made on (DATA-48); the 

128 file says so, and :func:`deserialize` does not need it back. 

129 """ 

130 document: dict[str, object] = {"schema": SCHEMA_VERSION} 

131 if dataset: 

132 document["dataset"] = str(dataset) 

133 document["annotations"] = store_to_records(store) 

134 return json.dumps(document, indent=2) 

135 

136 

137def file_dataset(text: str) -> str | None: 

138 """The ``dataset`` an annotations file names, if any (schema 3+).""" 

139 try: 

140 data = json.loads(text) 

141 except ValueError: 

142 return None 

143 name = data.get("dataset") if isinstance(data, dict) else None 

144 return name if isinstance(name, str) and name else None 

145 

146 

147class AnnotationsFileError(ValueError): 

148 """A JSON document that is not an annotations file — wrong shape, not syntax.""" 

149 

150 

151_ID_TYPES = (str, int, float) 

152 

153 

154def _check_record(index: int, record: object) -> None: 

155 """Raise :class:`AnnotationsFileError` unless ``record`` has the exported shape.""" 

156 where = f"entry {index + 1}" 

157 if not isinstance(record, dict): 

158 raise AnnotationsFileError(f"{where} is not an annotation") 

159 for field in ("participant_id", "trial_id"): 

160 value = record.get(field) 

161 if isinstance(value, bool) or not isinstance(value, _ID_TYPES): 

162 raise AnnotationsFileError(f"{where} has no usable {field}") 

163 screen_id = record.get("screen_id") 

164 if screen_id is not None and ( 

165 isinstance(screen_id, bool) or not isinstance(screen_id, _ID_TYPES) 

166 ): 

167 raise AnnotationsFileError(f"{where} has an unusable screen_id") 

168 if not isinstance(record.get("star", False), bool): 

169 raise AnnotationsFileError(f"{where}: star must be true or false") 

170 tags = record.get("tags", []) 

171 if tags is not None and ( 

172 not isinstance(tags, list) 

173 or any(isinstance(t, bool) or not isinstance(t, _ID_TYPES) for t in tags) 

174 ): 

175 raise AnnotationsFileError(f"{where}: tags must be a list of text labels") 

176 note = record.get("note", "") 

177 if note is not None and not isinstance(note, str): 

178 raise AnnotationsFileError(f"{where}: note must be text") 

179 

180 

181def deserialize(text: str) -> dict[Key, Entry]: 

182 """Parse a JSON document (object with ``annotations`` or a bare list). 

183 

184 The shape is checked before anything is built, so a JSON file that is not 

185 an annotations file — another app's, a settings file, a hand edit gone 

186 wrong — raises :class:`AnnotationsFileError` (a ``ValueError``) rather than 

187 importing nothing or failing half-way. Invalid JSON raises ``ValueError`` 

188 from :func:`json.loads`. 

189 """ 

190 data = json.loads(text) 

191 if isinstance(data, dict): 

192 if "annotations" not in data: 

193 raise AnnotationsFileError("it has no annotations list") 

194 records = data["annotations"] 

195 else: 

196 records = data 

197 if not isinstance(records, list): 

198 raise AnnotationsFileError("its annotations are not a list") 

199 for index, record in enumerate(records): 

200 _check_record(index, record) 

201 return records_to_store(records) 

202 

203 

204def _trial_of(key: Key) -> tuple[str, str]: 

205 return (key[0], key[1]) 

206 

207 

208def _trial_set(trials: Iterable[tuple[object, object]]) -> frozenset[tuple[str, str]]: 

209 if isinstance(trials, frozenset): 

210 return trials # already built by the caller, as strings 

211 return frozenset((str(pid), str(tid)) for pid, tid in trials) 

212 

213 

214def records_in(store: dict[Key, Entry], trials) -> list[dict]: 

215 """The records of ``store`` on the trials in ``trials``, sorted. 

216 

217 ``trials`` is ``(participant_id, trial_id)`` pairs. Keeps what an Export 

218 bundle writes to the trials it exports, and tells a dataset's annotations 

219 on trials it still has from ones it no longer has (DATA-48). A screen 

220 annotation belongs to its parent trial. 

221 """ 

222 keep = _trial_set(trials) 

223 return store_to_records( 

224 {key: entry for key, entry in store.items() if _trial_of(key) in keep} 

225 ) 

226 

227 

228def merge_records( 

229 store: dict[Key, Entry], records: list[dict], trials 

230) -> tuple[int, int]: 

231 """Add ``records`` on the trials in ``trials`` to ``store``, in place. 

232 

233 An imported entry replaces the one already on its key — the file is the 

234 newer statement about that trial — and a record for a trial the dataset 

235 does not have is skipped, as a backup restore skips what does not match 

236 (UX-174 r2). Returns ``(applied, skipped)``. 

237 """ 

238 keep = _trial_set(trials) 

239 incoming = records_to_store(records) 

240 # A file saved before composite ids escaped a `_` inside a part names those 

241 # trials by their old spelling; read it the dataset's way when that is 

242 # unambiguous (`data.respell_reading`). 

243 if any(_trial_of(key) not in keep for key in incoming): 

244 incoming = { 

245 ( 

246 key 

247 if _trial_of(key) in keep 

248 else (*respell_reading(key[0], key[1], keep), *key[2:]) 

249 ): entry 

250 for key, entry in incoming.items() 

251 } 

252 applied = {key: entry for key, entry in incoming.items() if _trial_of(key) in keep} 

253 store.update(applied) 

254 return len(applied), len(incoming) - len(applied) 

255 

256 

257def drop_records(store: dict[Key, Entry], records: list[dict]) -> int: 

258 """Remove the entries ``records`` name from ``store``. Returns how many.""" 

259 removed = 0 

260 for record in records: 

261 pid, tid = str(record["participant_id"]), str(record["trial_id"]) 

262 screen_id = record.get("screen_id") 

263 key = (pid, tid, str(screen_id)) if screen_id not in (None, "") else (pid, tid) 

264 if store.pop(key, None) is not None: 

265 removed += 1 

266 return removed 

267 

268 

269# --------------------------------------------------------------------------- 

270# Session-backed layer 

271# --------------------------------------------------------------------------- 

272 

273 

274def _store() -> dict[Key, Entry]: 

275 return st.session_state.setdefault(ANNOTATIONS_STATE_KEY, {}) 

276 

277 

278# --------------------------------------------------------------------------- 

279# DATA-48 — the annotations belong to a dataset 

280# 

281# `ANNOTATIONS_STATE_KEY` holds the selected dataset's store; every other 

282# dataset's waits in `DATASET_STORE_KEY`, and `activate_dataset` swaps them in 

283# and out when the selection changes — DATA-47's design for the metadata 

284# tables, taken over rather than reinvented, so the two follow one selection by 

285# one set of rules. A dataset component in every key was the alternative; it 

286# would have made each of the store's readers (the editor, the pickers' 

287# markers, the filters, the Data page, the Export bundle) pass a dataset, where 

288# the swap leaves them all reading the one store they already read. 

289# 

290# These functions take the session mapping explicitly, like `metadata`'s, so 

291# the swap is testable on a plain dict. 

292# --------------------------------------------------------------------------- 

293 

294#: Every dataset's annotations but the selected one's: ``{dataset: {key: 

295#: entry}}``. The selected dataset is never in it. 

296DATASET_STORE_KEY = "_annotations_by_dataset" 

297#: Which dataset :data:`ANNOTATIONS_STATE_KEY`'s store belongs to right now. 

298#: Absent until the first :func:`activate_dataset`. 

299OWNER_KEY = "_annotations_owner" 

300#: Entries that belong to no dataset yet — a pre-DATA-48 recovery cache's that 

301#: no upload's trials claimed (:func:`restore_payload`). The first *real* 

302#: dataset activated adopts them; the add wizard's never does, since nothing 

303#: of it is cached and they would be lost with it. 

304UNASSIGNED_KEY = "_annotations_unassigned" 

305#: Bumped on every change to the stored (not live) stores — the recovery 

306#: cache's cheap "did it change" test (:func:`store_signature`), the pattern 

307#: of ``metadata.STORE_REVISION_KEY``: serializing every dataset's annotations 

308#: on every rerun is not cheap once there are several thousand. 

309STORE_REVISION_KEY = "_annotations_store_revision" 

310#: The add-dataset wizard's dataset, before it has a name. The same token as 

311#: ``metadata.PENDING_DATASET`` (pinned by a test; not imported, since 

312#: `metadata` pulls in the data layer). Never cached. 

313PENDING_DATASET = "\x00pending" 

314#: CMP-8's key prefix for scanpath B's filters (``tabs._COMPARE_FILTER_PREFIX``) 

315#: and the picker's "same dataset" answer (``compare_source.THIS_DATASET``). 

316#: Pinned equal by a test; not imported, since both modules import this one. 

317_COMPARE_PREFIX = "cmp" 

318_COMPARE_SAME_DATASET = "This dataset" 

319 

320 

321def _stored(session) -> dict[str, dict[Key, Entry]]: 

322 store = session.get(DATASET_STORE_KEY) 

323 return dict(store) if isinstance(store, dict) else {} 

324 

325 

326def _live(session) -> dict[Key, Entry]: 

327 live = session.get(ANNOTATIONS_STATE_KEY) 

328 return dict(live) if isinstance(live, dict) else {} 

329 

330 

331def _unassigned(session) -> dict[Key, Entry]: 

332 pool = session.get(UNASSIGNED_KEY) 

333 return dict(pool) if isinstance(pool, dict) else {} 

334 

335 

336def _set_unassigned(session, pool: dict) -> None: 

337 if pool: 

338 session[UNASSIGNED_KEY] = pool 

339 else: 

340 session.pop(UNASSIGNED_KEY, None) 

341 

342 

343def _bump(session) -> None: 

344 session[STORE_REVISION_KEY] = int(session.get(STORE_REVISION_KEY) or 0) + 1 

345 

346 

347def _reseed_editors_in(session) -> None: 

348 """Drop the per-trial editors' widget state from ``session``. 

349 

350 Their keys carry the trial's ids, not the dataset's: left in place across a 

351 swap, the editor of a trial the next dataset shares would seed from the last 

352 dataset's values and write them into the new dataset's store. Nothing typed 

353 is lost by it — every editor field saves itself on change (``_save_entry``), 

354 and a widget's callback runs before the run that swaps. 

355 """ 

356 for key in [ 

357 k 

358 for k in list(session.keys()) 

359 if isinstance(k, str) and k.startswith(_WIDGET_PREFIX) 

360 ]: 

361 del session[key] 

362 

363 

364def activate_dataset(session, dataset: str, *, adopt: bool = True) -> bool: 

365 """Make ``dataset``'s annotations the session store; whether it changed. 

366 

367 Called by ``app.main`` on every run with the dataset on screen. When that 

368 changed, the outgoing dataset's store is filed away and the incoming one's 

369 comes back. A store with no owner yet (entries put there before the first 

370 run) joins the :data:`UNASSIGNED_KEY` pool, which :func:`adopt_unassigned` 

371 hands on. ``app.main`` passes ``adopt=False`` and adopts only 

372 once the load has said which dataset is really shown (a missing corpus is 

373 shown as the demo); the default adopts here, for callers with no load. 

374 """ 

375 dataset = str(dataset) 

376 owner = session.get(OWNER_KEY) 

377 if owner == dataset: 

378 if adopt: 

379 adopt_unassigned(session) 

380 return False 

381 store = _stored(session) 

382 live = _live(session) 

383 unassigned = _unassigned(session) 

384 if owner is None: 

385 unassigned = {**live, **unassigned} 

386 elif live: 

387 store[str(owner)] = live 

388 else: 

389 store.pop(str(owner), None) 

390 session[ANNOTATIONS_STATE_KEY] = store.pop(dataset, None) or {} 

391 session[DATASET_STORE_KEY] = store 

392 _set_unassigned(session, unassigned) 

393 session[OWNER_KEY] = dataset 

394 _bump(session) 

395 _reseed_editors_in(session) 

396 if adopt: 

397 adopt_unassigned(session) 

398 return True 

399 

400 

401def adopt_unassigned(session) -> int: 

402 """Give the :data:`UNASSIGNED_KEY` pool to the dataset on screen; how many. 

403 

404 The pool holds a pre-DATA-48 cache's entries that no restored upload's 

405 trials claimed (:func:`restore_payload`) — so they are on trials of a 

406 built-in or public corpus, and only such a dataset may adopt them. Never 

407 an upload (each was asked in the claim pass, and said no), and never the 

408 add wizard's unnamed dataset, which is not cached. The dataset's own entry 

409 wins a collision with an adopted one. 

410 """ 

411 pool = _unassigned(session) 

412 owner = session.get(OWNER_KEY) 

413 if not pool or owner is None or owner == PENDING_DATASET: 

414 return 0 

415 uploads = session.get("_datasets") 

416 if isinstance(uploads, dict) and owner in uploads: 

417 return 0 

418 session[ANNOTATIONS_STATE_KEY] = {**pool, **_live(session)} 

419 session.pop(UNASSIGNED_KEY, None) 

420 _bump(session) 

421 _reseed_editors_in(session) 

422 return len(pool) 

423 

424 

425def begin_pending_dataset(session) -> None: 

426 """Start the add-dataset wizard's dataset with no annotations of its own.""" 

427 store = _stored(session) 

428 if store.pop(PENDING_DATASET, None) is not None: 

429 session[DATASET_STORE_KEY] = store 

430 _bump(session) 

431 

432 

433def adopt_pending_dataset(session, dataset: str) -> None: 

434 """✅ Add dataset: the wizard's store becomes ``dataset``'s. 

435 

436 Only while the wizard's dataset is the selected one — relabelling another 

437 dataset's store as the new one's would move its annotations — and the new 

438 dataset starts clean: a stale store left under its name is dropped, not 

439 merged (``wizard._safe_dataset_name`` keeps such names from being chosen). 

440 """ 

441 begin_pending_dataset(session) 

442 if session.get(OWNER_KEY) != PENDING_DATASET: 

443 return 

444 dataset = str(dataset) 

445 store = _stored(session) 

446 if store.pop(dataset, None) is not None: 

447 session[DATASET_STORE_KEY] = store 

448 session[OWNER_KEY] = dataset 

449 _bump(session) 

450 

451 

452def forget_dataset(session, dataset: str) -> None: 

453 """A removed dataset's annotations go with it.""" 

454 dataset = str(dataset) 

455 store = _stored(session) 

456 if store.pop(dataset, None) is not None: 

457 session[DATASET_STORE_KEY] = store 

458 if session.get(OWNER_KEY) == dataset: 

459 session[ANNOTATIONS_STATE_KEY] = {} 

460 session.pop(OWNER_KEY, None) 

461 _reseed_editors_in(session) 

462 _bump(session) 

463 

464 

465def current_dataset(session) -> str | None: 

466 """The dataset the session store belongs to; ``None`` before the first run 

467 and while the add wizard's unnamed dataset is open.""" 

468 owner = session.get(OWNER_KEY) 

469 return None if owner in (None, PENDING_DATASET) else str(owner) 

470 

471 

472def dataset_names(session) -> set[str]: 

473 """Every dataset name that holds (or owns) an annotation store.""" 

474 names = {str(name) for name, entries in _stored(session).items() if entries} 

475 owner = session.get(OWNER_KEY) 

476 if owner is not None: 

477 names.add(str(owner)) 

478 names.discard(PENDING_DATASET) 

479 return names 

480 

481 

482def rename_dataset(session, old: str, new: str) -> bool: 

483 """A renamed dataset keeps its annotations; whether they moved. 

484 

485 Refuses — changing nothing — when ``new`` already holds a store of its own, 

486 which the rename would otherwise overwrite. ``wizard.rename_dataset`` never 

487 asks for such a name (``_safe_dataset_name`` avoids :func:`dataset_names`). 

488 """ 

489 old, new = str(old), str(new) 

490 if old == new: 

491 return True 

492 store = _stored(session) 

493 if new in store or session.get(OWNER_KEY) == new: 

494 return False 

495 if old in store: 

496 session[DATASET_STORE_KEY] = { 

497 (new if key == old else key): value for key, value in store.items() 

498 } 

499 if session.get(OWNER_KEY) == old: 

500 session[OWNER_KEY] = new 

501 _bump(session) 

502 return True 

503 

504 

505def store_for(session, dataset: str) -> dict[Key, Entry]: 

506 """``dataset``'s annotation store — the live one while it is selected.""" 

507 dataset = str(dataset) 

508 if session.get(OWNER_KEY) == dataset: 

509 return _live(session) 

510 return dict(_stored(session).get(dataset) or {}) 

511 

512 

513def store_for_prefix(prefix: str = "") -> dict[Key, Entry]: 

514 """The store the trial filters under key ``prefix`` narrow by (DATA-48). 

515 

516 The main pool's filters read the selected dataset's. Compare mode's 

517 scanpath B (the ``cmp`` prefix) can come from another dataset, and then its 

518 ⭐ / tag filters, its tag list and its picker's markers are *that* 

519 dataset's — the rule ``metadata.attached_for`` follows for its tables. 

520 Outside a script run (API, CLI) there is no store: ``{}``. 

521 """ 

522 try: 

523 session = st.session_state 

524 if prefix != _COMPARE_PREFIX: 

525 return _live(session) 

526 other = session.get(COMPARE_SOURCE_STATE_KEY) 

527 except Exception: # no script run context 

528 return {} 

529 if not other or other == _COMPARE_SAME_DATASET: 

530 return _live(session) 

531 return store_for(session, str(other)) 

532 

533 

534def upload_trials(entry) -> frozenset[tuple[str, str]]: 

535 """An upload's ``(participant, trial)`` pairs, as annotations are keyed. 

536 

537 The trial picker's ids: `utils.build_combo_options` lists the fixation 

538 table's trials under ``unique_trial_id`` when there is one, else 

539 ``trial_id``. A table without fixations is read from its words instead. 

540 """ 

541 for key in ("fixations", "words"): 

542 frame = entry.get(key) if isinstance(entry, dict) else None 

543 if not isinstance(frame, pd.DataFrame) or frame.empty: 

544 continue 

545 trial_col = "unique_trial_id" if "unique_trial_id" in frame else "trial_id" 

546 if {"participant_id", trial_col} <= set(frame.columns): 

547 pairs = frame[["participant_id", trial_col]].drop_duplicates() 

548 return frozenset( 

549 zip(pairs["participant_id"].astype(str), pairs[trial_col].astype(str)) 

550 ) 

551 return frozenset() 

552 

553 

554def dataset_records(session) -> dict[str, list[dict]]: 

555 """Every dataset's annotations, the selected one's live: ``{name: records}``. 

556 

557 What the recovery cache writes. The wizard's unnamed dataset is left out, 

558 and so are entries no dataset owns yet — :func:`unassigned_records`. 

559 """ 

560 stores = _stored(session) 

561 owner = session.get(OWNER_KEY) 

562 if owner is not None: 

563 stores[str(owner)] = _live(session) 

564 stores.pop(PENDING_DATASET, None) 

565 return { 

566 name: store_to_records(entries) 

567 for name, entries in sorted(stores.items()) 

568 if entries 

569 } 

570 

571 

572def unassigned_records(session) -> list[dict]: 

573 """Entries no dataset owns yet — cached as such, so none is lost. 

574 

575 The :data:`UNASSIGNED_KEY` pool, plus the session store itself while no 

576 dataset owns it (a run that returned before :func:`activate_dataset`). 

577 """ 

578 pool = _unassigned(session) 

579 if session.get(OWNER_KEY) is None: 

580 pool = {**_live(session), **pool} 

581 return store_to_records(pool) 

582 

583 

584def cache_payload(session) -> dict: 

585 """The recovery cache's ``annotations`` value: ``{"datasets": …}`` (DATA-48).""" 

586 payload: dict[str, object] = {"datasets": dataset_records(session)} 

587 unassigned = unassigned_records(session) 

588 if unassigned: 

589 payload["unassigned"] = unassigned 

590 return payload 

591 

592 

593def store_signature(session) -> list: 

594 """A cheap fingerprint of every dataset's annotations, for the cache. 

595 

596 The live store by content (the editor writes it directly), the rest by 

597 :data:`STORE_REVISION_KEY` — ``metadata.store_signature``'s pattern — so a 

598 rerun does not serialize every stored dataset's annotations. 

599 """ 

600 return [ 

601 int(session.get(STORE_REVISION_KEY) or 0), 

602 str(session.get(OWNER_KEY)), 

603 store_to_records(_live(session)), 

604 store_to_records(_unassigned(session)), 

605 ] 

606 

607 

608def _valid_records(records) -> list[dict]: 

609 """One malformed record costs that record, not the rest.""" 

610 return [ 

611 record 

612 for record in (records if isinstance(records, list) else []) 

613 if isinstance(record, dict) 

614 ] 

615 

616 

617def restore_payload(session, payload) -> int: 

618 """Put back what :func:`cache_payload` wrote; how many annotations landed. 

619 

620 A dataset this session already holds annotations for keeps its own — the 

621 restore's ``setdefault`` rule. 

622 

623 **The one-time migration.** A manifest written before DATA-48 stored one 

624 flat list of records naming no dataset; it is read as ``unassigned``. Each 

625 entry goes, first, to every upload whose trials include it — the uploads' 

626 frames are back in ``_datasets`` by now (``persistence.restore_state`` 

627 restores them first), and an annotation on a trial only one dataset has is 

628 that dataset's. An entry no upload claims (one on a built-in or public 

629 corpus, which is not in memory to ask) waits in :data:`UNASSIGNED_KEY` for 

630 the first real dataset the session opens — the one the manifest had 

631 selected, unless a link names another. Nothing is dropped, and the next save 

632 writes everything back per dataset, so the conversion happens once. 

633 """ 

634 if isinstance(payload, list): 

635 datasets, unassigned = {}, payload 

636 elif isinstance(payload, dict): 

637 datasets = payload.get("datasets") 

638 datasets = datasets if isinstance(datasets, dict) else {} 

639 unassigned = payload.get("unassigned") or [] 

640 else: 

641 datasets, unassigned = {}, [] 

642 live = _live(session) 

643 owner = session.get(OWNER_KEY) 

644 store = _stored(session) 

645 restored = 0 

646 

647 def file_under(name: str, entries: dict) -> None: 

648 nonlocal live 

649 if name == owner: 

650 live = {**entries, **live} 

651 else: 

652 store[name] = {**entries, **(store.get(name) or {})} 

653 

654 for name, records in datasets.items(): 

655 name = str(name) 

656 entries = records_to_store(_valid_records(records)) 

657 if not entries or name == PENDING_DATASET or name in store: 

658 continue 

659 if name == owner and live: 

660 continue 

661 file_under(name, entries) 

662 restored += len(entries) 

663 

664 legacy = records_to_store(_valid_records(unassigned)) 

665 if legacy: 

666 claimed: set[Key] = set() 

667 uploads = session.get("_datasets") 

668 for name, entry in (uploads if isinstance(uploads, dict) else {}).items(): 

669 trials = upload_trials(entry) 

670 mine = {k: e for k, e in legacy.items() if _trial_of(k) in trials} 

671 if mine: 

672 file_under(str(name), mine) 

673 claimed.update(mine) 

674 rest = {k: e for k, e in legacy.items() if k not in claimed} 

675 _set_unassigned(session, {**rest, **_unassigned(session)}) 

676 restored += len(legacy) 

677 session[ANNOTATIONS_STATE_KEY] = live 

678 if store: 

679 session[DATASET_STORE_KEY] = store 

680 _bump(session) 

681 return restored 

682 

683 

684def payload_count(payload) -> int: 

685 """How many annotations a cached ``annotations`` value holds, either shape.""" 

686 if isinstance(payload, list): 

687 return len(payload) 

688 if not isinstance(payload, dict): 

689 return 0 

690 datasets = payload.get("datasets") 

691 total = sum( 

692 len(records) 

693 for records in (datasets.values() if isinstance(datasets, dict) else []) 

694 if isinstance(records, list) 

695 ) 

696 unassigned = payload.get("unassigned") 

697 return total + (len(unassigned) if isinstance(unassigned, list) else 0) 

698 

699 

700def get_entry( 

701 participant_id: str, trial_id: str, screen_id: str | None = None 

702) -> Entry: 

703 key = ( 

704 (str(participant_id), str(trial_id), str(screen_id)) 

705 if screen_id not in (None, "") 

706 else (str(participant_id), str(trial_id)) 

707 ) 

708 return _store().get(key, default_entry()) 

709 

710 

711def set_entry( 

712 participant_id: str, 

713 trial_id: str, 

714 *, 

715 star: bool, 

716 tags: list[str], 

717 note: str, 

718 screen_id: str | None = None, 

719) -> None: 

720 """Upsert an annotation; empty entries are pruned to keep the store small.""" 

721 key = ( 

722 (str(participant_id), str(trial_id), str(screen_id)) 

723 if screen_id not in (None, "") 

724 else (str(participant_id), str(trial_id)) 

725 ) 

726 entry = _normalize_entry(star, tags, note) 

727 store = _store() 

728 if is_empty_entry(entry): 

729 store.pop(key, None) 

730 else: 

731 store[key] = entry 

732 

733 

734def known_tags(prefix: str = "", *, trial_level: bool = False) -> list[str]: 

735 """Preset tags plus any tag used in the dataset's store, sorted. 

736 

737 ``prefix`` picks the dataset as :func:`store_for_prefix` does, so compare 

738 mode's scanpath B lists its own dataset's tags (DATA-48). 

739 

740 ``trial_level`` leaves out tags used only on screen annotations — what the 

741 trial filters offer, since they read the trial's own entry (:func:`select_keys`) 

742 and a screen-only tag there could never match. 

743 """ 

744 tags: set[str] = set(PRESET_TAGS) 

745 store = store_for_prefix(prefix) if prefix else _store() 

746 for key, entry in store.items(): 

747 if trial_level and len(key) > 2: 

748 continue 

749 tags.update(entry.get("tags", [])) 

750 return sorted(tags) 

751 

752 

753def has_screen_annotations(prefix: str = "") -> bool: 

754 """Whether the dataset's store holds any screen annotation.""" 

755 store = store_for_prefix(prefix) if prefix else _store() 

756 return any(len(key) > 2 for key in store) 

757 

758 

759def current_records() -> list[dict]: 

760 """The open dataset's annotations as records — the Export bundle's (UX-179).""" 

761 return store_to_records(_store()) 

762 

763 

764def _reseed_trial_editors() -> None: 

765 """Drop the per-trial editors' widget state, so they re-seed from the store.""" 

766 _reseed_editors_in(st.session_state) 

767 

768 

769# --------------------------------------------------------------------------- 

770# UI render helpers 

771# --------------------------------------------------------------------------- 

772 

773 

774def _add_tag_callback(tags_key: str, newtag_key: str, save_args: tuple) -> None: 

775 """on_change for the 'add tag' input: append to the multiselect's state. 

776 

777 Runs before the next rerun, so writing the multiselect's session_state here 

778 is allowed (the widget hasn't been instantiated yet that run). Saves the 

779 entry too, like every other editor field (see :func:`_save_entry_callback`).""" 

780 new_tag = str(st.session_state.get(newtag_key, "")).strip() 

781 if not new_tag: 

782 return 

783 current = list(st.session_state.get(tags_key, [])) 

784 if new_tag not in current: 

785 st.session_state[tags_key] = current + [new_tag] 

786 st.session_state[newtag_key] = "" 

787 _save_entry_callback(*save_args) 

788 

789 

790def _save_entry_callback( 

791 participant_id: str, 

792 trial_id: str, 

793 screen_id: str | None, 

794 star_key: str, 

795 tags_key: str, 

796 note_key: str, 

797) -> None: 

798 """Persist the editor's fields as soon as one changes, before the rerun. 

799 

800 Widget callbacks run before the app body. Without this the picker reads the 

801 previous annotation store, then ``render_trial_annotations`` saves the new 

802 value later in the run — making the star marker appear one click behind the 

803 checkbox. And (DATA-48) the run that switches dataset drops the editors' 

804 widget state in ``activate_dataset`` before any editor renders: a tag or a 

805 note saved only by the render body would be lost with it. Saved here, it is 

806 already in the outgoing dataset's store when the swap files that away. 

807 """ 

808 entry = get_entry(participant_id, trial_id, screen_id) 

809 state = st.session_state 

810 set_entry( 

811 participant_id, 

812 trial_id, 

813 star=bool(state[star_key]) if star_key in state else bool(entry["star"]), 

814 tags=list(state[tags_key]) if tags_key in state else list(entry["tags"]), 

815 note=str(state[note_key]) if note_key in state else str(entry["note"]), 

816 screen_id=screen_id, 

817 ) 

818 

819 

820def widget_slug( 

821 participant_id: str, trial_id: str, screen_id: str | None = None 

822) -> str: 

823 """The per-trial editor's widget-key suffix: the store key, unambiguously. 

824 

825 BUG-110: joining the ids with ``__`` and writing the parent scope as the 

826 word ``parent`` gave ``("a__b", "c")`` and ``("a", "b__c")`` — or a trial's 

827 parent and its screen named ``parent`` — the same widget keys, so opening 

828 the one seeded its editor from the other's values and saved them as its 

829 own. A JSON list keeps every id whole and writes the parent as ``null``. 

830 """ 

831 ident = [str(participant_id), str(trial_id)] 

832 ident.append(None if screen_id is None else str(screen_id)) 

833 return json.dumps(ident, ensure_ascii=False) 

834 

835 

836def render_trial_annotations( 

837 participant_id: str, 

838 trial_id: str, 

839 *, 

840 screen_id: str | None = None, 

841 bare: bool = False, 

842) -> None: 

843 """Render the per-trial annotations (star / tags / notes). 

844 

845 ``bare=True`` drops the expander wrapper so it can sit inside a subtab.""" 

846 annotation_screen = None 

847 if screen_id is not None: 

848 scope_key = f"{_WIDGET_PREFIX}scope_{widget_slug(participant_id, trial_id)}" 

849 scope = panel_field( 

850 st, 

851 "radio", 

852 "Annotation scope", 

853 options=["Parent trial", "This screen"], 

854 # #374: the stored option keeps its name; "parent" is internal. 

855 format_func=lambda option: ( 

856 "Whole trial" if option == "Parent trial" else option 

857 ), 

858 key=scope_key, 

859 horizontal=True, 

860 help="Whole trial: every screen of this trial. This screen: only the " 

861 "screen on view.", 

862 ) 

863 if scope == "This screen": 

864 annotation_screen = str(screen_id) 

865 entry = get_entry(participant_id, trial_id, annotation_screen) 

866 slug = widget_slug(participant_id, trial_id, annotation_screen) 

867 star_key = f"{_WIDGET_PREFIX}star_{slug}" 

868 tags_key = f"{_WIDGET_PREFIX}tags_{slug}" 

869 note_key = f"{_WIDGET_PREFIX}note_{slug}" 

870 newtag_key = f"{_WIDGET_PREFIX}newtag_{slug}" 

871 save_args = ( 

872 participant_id, 

873 trial_id, 

874 annotation_screen, 

875 star_key, 

876 tags_key, 

877 note_key, 

878 ) 

879 

880 # Seed widget state once from the store (re-seeds after a JSON import, which 

881 # clears these keys). 

882 st.session_state.setdefault(star_key, entry["star"]) 

883 st.session_state.setdefault(tags_key, list(entry["tags"])) 

884 st.session_state.setdefault(note_key, entry["note"]) 

885 

886 label = f"{ICONS['annotations']} Annotations" + ( 

887 f" {ICONS['favorite']}" if entry["star"] else "" 

888 ) 

889 if entry["tags"]: 

890 label += f" · {', '.join(entry['tags'])}" 

891 container = st.container() if bare else st.expander(label, expanded=False) 

892 with container: 

893 # UX-69: `label | field` rows, so the five stacked controls fit under the 

894 # plot without scrolling. Unlike the rail (`controls._labeled`'s note), 

895 # the checkbox is split too: here the label column is the row spine every 

896 # other field lines up on, and a native checkbox would put its box — 

897 # not its title — at that edge. 

898 star = panel_field( 

899 st, 

900 "checkbox", 

901 f"{ICONS['favorite']} Favorite (star this trial)", 

902 display=f"{ICONS['favorite']} Favorite", 

903 key=star_key, 

904 help="Star this trial (with scope *This screen*, this screen only).", 

905 on_change=_save_entry_callback, 

906 args=save_args, 

907 ) 

908 # Options must include every currently-selected tag (incl. ones added 

909 # via the input) or st.multiselect raises. 

910 options = sorted( 

911 set(known_tags()) 

912 | set(entry["tags"]) 

913 | set(st.session_state.get(tags_key, [])) 

914 ) 

915 tags_help = "Select tags or add one." 

916 label_col, tags_col, add_col = st.columns( 

917 [PANEL_LABEL_W, 0.52, 0.28], 

918 gap="xsmall", 

919 vertical_alignment="center", 

920 ) 

921 row_label(label_col, "Tags", tags_help) 

922 # ENG-49: built directly in a column, so it does not go through 

923 # `fields.labeled` and needs its own `wrap` — a trial's tags are the 

924 # thing this row exists to show, and 1.63 would otherwise scroll all 

925 # but the first one or two out of sight. 

926 tags = tags_col.multiselect( 

927 "Tags", 

928 options=options, 

929 key=tags_key, 

930 help=tags_help, 

931 label_visibility="collapsed", 

932 wrap=True, 

933 on_change=_save_entry_callback, 

934 args=save_args, 

935 ) 

936 add_col.text_input( 

937 "Add a tag", 

938 key=newtag_key, 

939 placeholder="Add a new tag", 

940 on_change=_add_tag_callback, 

941 args=(tags_key, newtag_key, save_args), 

942 label_visibility="collapsed", 

943 ) 

944 # Top-aligned: the box is ~100px tall, and a centred title would float 

945 # opposite the middle of an empty note rather than beside its first line. 

946 note = panel_field( 

947 st, 

948 "text_area", 

949 "Notes", 

950 align="top", 

951 key=note_key, 

952 placeholder="Researcher notes for this trial…", 

953 height=100, 

954 on_change=_save_entry_callback, 

955 args=save_args, 

956 ) 

957 set_entry( 

958 participant_id, 

959 trial_id, 

960 star=star, 

961 tags=tags, 

962 note=note, 

963 screen_id=annotation_screen, 

964 ) 

965 # UX-76: no shortcut button under this — the caption names where the 

966 # whole dataset's annotations are listed instead. 

967 st.caption( 

968 "Every annotation on this dataset is listed on " 

969 f"{ICONS['view_data']} **Data Management → Annotations**, to export, import " 

970 "or delete." 

971 ) 

972 

973 

974def select_keys( 

975 store: dict[Key, Entry], 

976 keys: list[Key], 

977 *, 

978 favorites_only: bool = False, 

979 required_tags: list[str] | None = None, 

980 excluded_tags: list[str] | None = None, 

981) -> list[Key]: 

982 """Pure core of :func:`filter_keys` — filter ``keys`` against ``store``. 

983 

984 Trial level only: each key is looked up as given, so a parent 

985 ``(participant_id, trial_id)`` key reads the trial's own annotation and 

986 never a screen's. That is the filters' stated scope — a screen star or tag 

987 neither keeps nor drops its trial, and 

988 the panel offers only trial-level tags (:func:`known_tags`). 

989 

990 - ``favorites_only``: keep only starred trials. 

991 - ``required_tags``: keep trials carrying *any* of these tags. 

992 - ``excluded_tags``: drop trials carrying *any* of these tags. 

993 """ 

994 required = set(required_tags or []) 

995 excluded = set(excluded_tags or []) 

996 out: list[Key] = [] 

997 for key in keys: 

998 entry = store.get(key) 

999 tags = set(entry.get("tags", [])) if entry else set() 

1000 starred = bool(entry.get("star")) if entry else False 

1001 if favorites_only and not starred: 

1002 continue 

1003 if required and not (tags & required): 

1004 continue 

1005 if excluded and (tags & excluded): 

1006 continue 

1007 out.append(key) 

1008 return out 

1009 

1010 

1011def filter_keys( 

1012 keys: list[Key], 

1013 *, 

1014 favorites_only: bool = False, 

1015 required_tags: list[str] | None = None, 

1016 excluded_tags: list[str] | None = None, 

1017 prefix: str = "", 

1018) -> list[Key]: 

1019 """Session-backed wrapper around :func:`select_keys`. 

1020 

1021 ``prefix`` picks the dataset's store as :func:`store_for_prefix` does. 

1022 """ 

1023 return select_keys( 

1024 store_for_prefix(prefix) if prefix else _store(), 

1025 keys, 

1026 favorites_only=favorites_only, 

1027 required_tags=required_tags, 

1028 excluded_tags=excluded_tags, 

1029 ) 

1030 

1031 

1032# --------------------------------------------------------------------------- 

1033# 🗂️ Data → Annotations (UX-174 r2) 

1034# --------------------------------------------------------------------------- 

1035 

1036#: The dataset tab's widget keys. The nonce is bumped after an import or a 

1037#: delete, which gives the uploader and the table fresh keys: an uploader keeps 

1038#: its file (and would import it again on every rerun), and a table's row 

1039#: selection is by position, so it would point at other rows once some are gone. 

1040_DATASET_NONCE_KEY = "_dataset_annotations_nonce" 

1041_DATASET_NOTE_KEY = "_dataset_annotations_note" 

1042 

1043 

1044def _dataset_widget_key(name: str) -> str: 

1045 return f"dataset_annotations_{name}_{st.session_state.get(_DATASET_NONCE_KEY, 0)}" 

1046 

1047 

1048def _refresh_dataset_widgets(note: str) -> None: 

1049 st.session_state[_DATASET_NONCE_KEY] = ( 

1050 int(st.session_state.get(_DATASET_NONCE_KEY, 0)) + 1 

1051 ) 

1052 st.session_state[_DATASET_NOTE_KEY] = note 

1053 _reseed_trial_editors() 

1054 

1055 

1056def _plural(count: int, noun: str) -> str: 

1057 return f"{count:,} {noun}{'' if count == 1 else 's'}" 

1058 

1059 

1060def _import_dataset_annotations( 

1061 uploader_key: str, trials: frozenset, dataset_name: str = "" 

1062) -> None: 

1063 upload = st.session_state.get(uploader_key) 

1064 if upload is None: 

1065 return 

1066 try: 

1067 text = upload.getvalue().decode("utf-8") 

1068 records = store_to_records(deserialize(text)) 

1069 except (UnicodeDecodeError, ValueError) as exc: 

1070 # Nothing was merged yet, so the dataset's annotations are untouched, 

1071 # and the fresh uploader key below lets another file be chosen. 

1072 reason = f" ({exc})" if isinstance(exc, AnnotationsFileError) else "" 

1073 st.session_state[_DATASET_NOTE_KEY] = ( 

1074 f"error:That file is not an annotations JSON file{reason}. " 

1075 "Nothing was imported." 

1076 ) 

1077 st.session_state[_DATASET_NONCE_KEY] = ( 

1078 int(st.session_state.get(_DATASET_NONCE_KEY, 0)) + 1 

1079 ) 

1080 return 

1081 # DATA-48: into the open dataset, whatever the file names — a file from 

1082 # before annotations were per dataset names none, and one exported from 

1083 # another dataset is still the user's call to bring here. The name is said. 

1084 applied, skipped = merge_records(_store(), records, trials) 

1085 note = f"Imported {_plural(applied, 'annotation')}." 

1086 source = file_dataset(text) 

1087 if source and source != dataset_name: 

1088 note = f"Imported {_plural(applied, 'annotation')} exported from **{source}**." 

1089 if skipped: 

1090 note += ( 

1091 f" Skipped {_plural(skipped, 'annotation')} on trials this dataset " 

1092 "doesn't have." 

1093 ) 

1094 _refresh_dataset_widgets(note) 

1095 

1096 

1097#: Cell text of a row's **Open** button — `ButtonColumn` takes its label from 

1098#: the cell value, as `tabs._OPEN_TRIAL_LABEL` does. 

1099_OPEN_LABEL = f"{ICONS['open']} Open" 

1100 

1101 

1102def _open_annotation( 

1103 click_key: str, records: list[dict], trials: frozenset, open_trials: frozenset 

1104) -> None: 

1105 """A row's **Open**: show its reading (and screen) in the Scanpath view. 

1106 

1107 The click is a callback, so it parks the request with 

1108 ``url_state.request_trial`` — the Corpus Analysis tables' hop. It opens 

1109 only that exact reading: one the dataset hasn't loaded, or one the trial 

1110 filters hide, is explained here instead, since the picker would otherwise 

1111 land on another reader's trial of the same id or stay where it was. 

1112 """ 

1113 click = st.session_state.get(click_key) 

1114 row = click.get("row") if isinstance(click, dict) else None 

1115 if row is None or not 0 <= row < len(records): 

1116 return 

1117 record = records[row] 

1118 pid, tid = str(record["participant_id"]), str(record["trial_id"]) 

1119 where = f"participant **{pid}**, trial **{tid}**" 

1120 if (pid, tid) not in trials: 

1121 st.session_state[_DATASET_NOTE_KEY] = ( 

1122 f"error:Can't open {where}: this dataset hasn't loaded that trial." 

1123 ) 

1124 return 

1125 if (pid, tid) not in open_trials: 

1126 st.session_state[_DATASET_NOTE_KEY] = ( 

1127 f"error:Can't open {where}: the trial filters hide it. Clear or " 

1128 f"change the filters on {ICONS['view_scanpath']} **Scanpath**, " 

1129 "then open it again." 

1130 ) 

1131 return 

1132 # Imported at call time: `url_state` imports the controls, which import 

1133 # this module. 

1134 from .url_state import request_trial 

1135 

1136 request_trial(pid, tid, screen_id=record.get("screen_id")) 

1137 

1138 

1139def _delete_dataset_annotations(records: list[dict]) -> None: 

1140 removed = drop_records(_store(), records) 

1141 _refresh_dataset_widgets(f"Deleted {_plural(removed, 'annotation')}.") 

1142 

1143 

1144def _annotations_frame( 

1145 records: list[dict], trials: frozenset, trial_labels=None 

1146) -> pd.DataFrame: 

1147 """The Annotations table. ``trial_labels`` maps a trial id to the trial 

1148 picker's label for it (#374 F5: one trial label everywhere).""" 

1149 shown = trial_labels or {} 

1150 frame = pd.DataFrame( 

1151 { 

1152 "Participant": [r["participant_id"] for r in records], 

1153 "Trial": [ 

1154 shown.get(str(r["trial_id"]), str(r["trial_id"])) for r in records 

1155 ], 

1156 "Screen": [r.get("screen_id", "") for r in records], 

1157 "Favorite": [r["star"] for r in records], 

1158 "Tags": [r["tags"] for r in records], 

1159 "Note": [r["note"] for r in records], 

1160 "In dataset": [ 

1161 (str(r["participant_id"]), str(r["trial_id"])) in trials 

1162 for r in records 

1163 ], 

1164 } 

1165 ) 

1166 if not frame["Screen"].astype(bool).any(): 

1167 frame = frame.drop(columns="Screen") 

1168 if frame["In dataset"].all(): 

1169 frame = frame.drop(columns="In dataset") 

1170 return frame 

1171 

1172 

1173def render_dataset_annotations( 

1174 trials, *, dataset_name: str, open_trials=None, trial_labels=None 

1175) -> None: 

1176 """🗂️ Data → **Annotations**: every annotation the open dataset holds. 

1177 

1178 One table — participant, trial, favorite, tags, note — with **Export** (this 

1179 dataset's annotations as JSON), **Import** (the same file, or the 

1180 ``annotations.json`` of an Export bundle: entries on trials the dataset has 

1181 are added, the rest are skipped) 

1182 and **Delete**, for the rows ticked in the table. Annotations are still made 

1183 per trial, on 🗺️ Scanpath → Annotations; this is where they are seen whole. 

1184 

1185 DATA-48: the session store *is* the dataset's now, so the table lists all 

1186 of it — including any entry on a trial the dataset has not loaded: one made 

1187 under another selection of a corpus' parts, or one a recovery cache from 

1188 before annotations were per dataset assigned here (:func:`restore_payload`). 

1189 Those are flagged rather than hidden, so an entry is never out of reach: 

1190 exported from here, it imports into the dataset it belongs to. 

1191 

1192 ``open_trials`` — the trials the Scanpath picker can show, after the trial 

1193 filters — gives each row an **Open** button (:func:`_open_annotation`); 

1194 without it the table has none. ``trial_labels`` writes each trial as 

1195 the trial picker does. 

1196 """ 

1197 trials = _trial_set(trials) 

1198 records = store_to_records(_store()) 

1199 note = st.session_state.pop(_DATASET_NOTE_KEY, None) 

1200 if note and note.startswith("error:"): 

1201 st.error(note.removeprefix("error:"), icon=ICONS["error"]) 

1202 elif note: 

1203 st.success(note, icon=ICONS["confirm"]) 

1204 

1205 toolbar = st.container( 

1206 key="dataset_annotations_toolbar", 

1207 horizontal=True, 

1208 vertical_alignment="center", 

1209 gap="small", 

1210 ) 

1211 toolbar.download_button( 

1212 "Export", 

1213 icon=ICONS["download"], 

1214 data=serialize(records_to_store(records), dataset=dataset_name), 

1215 file_name=f"{_file_slug(dataset_name)}_annotations.json", 

1216 mime="application/json", 

1217 key="dataset_annotations_export", 

1218 disabled=not records, 

1219 help="Download this dataset's annotations as a JSON file.", 

1220 ) 

1221 with toolbar.popover("Import", icon=ICONS["upload"]): 

1222 uploader_key = _dataset_widget_key("import") 

1223 st.file_uploader( 

1224 "Annotations file (JSON)", 

1225 type=["json"], 

1226 key=uploader_key, 

1227 on_change=_import_dataset_annotations, 

1228 args=(uploader_key, trials, dataset_name), 

1229 max_upload_size=upload_limit_mb(), 

1230 ) 

1231 st.caption( 

1232 "A file exported here or in an Export bundle. Annotations on trials " 

1233 "this dataset doesn't have are skipped, and an imported one replaces " 

1234 "what its trial already had." 

1235 ) 

1236 delete_slot = toolbar.container(width="content") 

1237 

1238 if not records: 

1239 st.caption( 

1240 "No annotations on this dataset yet. Star, tag or note a trial on " 

1241 f"{ICONS['view_scanpath']} **Scanpath → Annotations**." 

1242 ) 

1243 return 

1244 elsewhere = len(records) - len(records_in(_store(), trials)) 

1245 if elsewhere: 

1246 st.caption( 

1247 f"{_plural(elsewhere, 'annotation')} here " 

1248 f"{'is' if elsewhere == 1 else 'are'} on trials this dataset hasn't " 

1249 "loaded (*In dataset* unticked). They stay with this " 

1250 "dataset; to move them to another, **Export** them here and " 

1251 "**Import** them there." 

1252 ) 

1253 frame = _annotations_frame(records, trials, trial_labels) 

1254 column_config = {} 

1255 if open_trials is not None: 

1256 frame.insert(0, "Open", _OPEN_LABEL) 

1257 open_key = _dataset_widget_key("open") 

1258 column_config["Open"] = st.column_config.ButtonColumn( 

1259 "", 

1260 type="tertiary", 

1261 width="small", 

1262 help="Show this trial — and its screen, for a screen annotation — " 

1263 "in the Scanpath view.", 

1264 on_click=_open_annotation, 

1265 args=(open_key, records, trials, _trial_set(open_trials)), 

1266 key=open_key, 

1267 ) 

1268 event = st.dataframe( 

1269 frame, 

1270 hide_index=True, 

1271 width="stretch", 

1272 on_select="rerun", 

1273 selection_mode="multi-row", 

1274 key=_dataset_widget_key("table"), 

1275 column_config={ 

1276 **column_config, 

1277 "Favorite": st.column_config.CheckboxColumn("Favorite", width="small"), 

1278 "Tags": st.column_config.ListColumn("Tags"), 

1279 "Note": st.column_config.TextColumn("Note", width="large"), 

1280 "In dataset": st.column_config.CheckboxColumn( 

1281 "In dataset", 

1282 width="small", 

1283 help="Whether this dataset has the annotated trial.", 

1284 ), 

1285 }, 

1286 ) 

1287 picked = [records[i] for i in event.selection.rows if i < len(records)] 

1288 with delete_slot.popover( 

1289 f"Delete ({len(picked)})" if picked else "Delete", 

1290 icon=ICONS["delete"], 

1291 disabled=not picked, 

1292 help="Tick rows in the table to delete them.", 

1293 ): 

1294 st.write( 

1295 f"Delete {_plural(len(picked), 'annotation')}? This cannot be undone " 

1296 "— **Export** first to keep a copy." 

1297 ) 

1298 st.button( 

1299 "Delete", 

1300 icon=ICONS["delete"], 

1301 key="dataset_annotations_delete", 

1302 type="primary", 

1303 on_click=_delete_dataset_annotations, 

1304 args=(picked,), 

1305 ) 

1306 

1307 

1308def _file_slug(name: str) -> str: 

1309 slug = "".join(ch if ch.isalnum() else "_" for ch in str(name)).strip("_") 

1310 return slug.lower() or "dataset"