Coverage for scanpath_studio/annotations.py: 94%
514 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Per-trial researcher annotations: favorites (stars), tags, and free notes.
3Annotations are keyed by ``(participant_id, trial_id)`` and live in Streamlit
4session state, so they persist across reruns within a session. There is no
5backend; a local run keeps them in the recovery cache, and to share them 🗂️
6Data → **Annotations** exports and imports a dataset's as JSON (UX-174). The
7Scanpath view's Export bundle can include the same file (UX-179).
9**DATA-48 — annotations belong to a dataset**, as the metadata tables do since
10DATA-47: :data:`ANNOTATIONS_STATE_KEY` holds the *selected* dataset's store only
11(so every reader — the per-trial editor, the pickers' markers, the ⭐ / tag
12filters, the Data page, the Export bundle — reads it unchanged), and every other
13dataset's waits in :data:`DATASET_STORE_KEY`. :func:`activate_dataset`, called
14by ``app.main`` beside ``metadata.activate_dataset``, swaps them when the
15selection changes. Two datasets that reuse ``(participant, trial)`` ids
16therefore no longer share a star, a tag or a note.
18The module is split into a *pure* core (``records_to_store`` /
19``store_to_records`` / ``serialize`` / ``deserialize`` — no Streamlit, unit
20tested) and a thin session-backed layer plus the small render helpers used by
21``tabs.py`` and ``app.py``.
22"""
24from __future__ import annotations
26import json
27from collections.abc import Iterable
29import pandas as pd
30import streamlit as st
32from .constants import ICONS, upload_limit_mb
33from .data import respell_reading
34from .fields import PANEL_LABEL_W, panel_field, row_label
35from .session_keys import COMPARE_SOURCE_STATE_KEY
37ANNOTATIONS_STATE_KEY = "trial_annotations"
38#: The annotations file (🗂️ Data → Annotations, an Export bundle's
39#: ``annotations.json``). Schema 3 (DATA-48) adds the optional ``dataset`` the
40#: file was exported from — for the reader, not the importer: a file of any
41#: schema imports into the dataset that is open, the only one it can belong to.
42SCHEMA_VERSION = 3
44# Per-trial annotation widgets use this prefix so they can be cleared on import
45# (forcing a re-seed from the freshly loaded store), and on a dataset swap.
46_WIDGET_PREFIX = "annotrial_"
48# Always-available tag suggestions (users can add their own on top).
49PRESET_TAGS = ["To exclude", "Review", "Good example", "Check alignment"]
51# Parent entries use ``(participant_id, trial_id)``; screen entries add a third
52# ``screen_id`` component. Keeping parent keys unchanged preserves filtering and
53# every schema-1 annotation sidecar.
54Key = tuple[str, ...]
55Entry = dict[str, object]
58# ---------------------------------------------------------------------------
59# Pure core (no Streamlit) — unit tested in tests/test_annotations.py
60# ---------------------------------------------------------------------------
63def default_entry() -> Entry:
64 return {"star": False, "tags": [], "note": ""}
67def _normalize_entry(star: object, tags: object, note: object) -> Entry:
68 # A recovery-cache record is not checked like an imported file is
69 # (`deserialize`), so a stray scalar here is no tags rather than a crash.
70 tags = tags if isinstance(tags, (list, tuple, set, frozenset)) else []
71 clean_tags = sorted({str(t).strip() for t in tags if str(t).strip()})
72 return {"star": bool(star), "tags": clean_tags, "note": str(note or "").strip()}
75def is_empty_entry(entry: Entry) -> bool:
76 """True when an entry carries no information (and can be dropped)."""
77 return (
78 not entry.get("star")
79 and not entry.get("tags")
80 and not str(entry.get("note") or "").strip()
81 )
84def records_to_store(records: list[dict]) -> dict[Key, Entry]:
85 """Build a ``{(pid, tid): entry}`` store from a list of flat records."""
86 store: dict[Key, Entry] = {}
87 for rec in records or []:
88 pid = rec.get("participant_id")
89 tid = rec.get("trial_id")
90 if pid is None or tid is None:
91 continue
92 entry = _normalize_entry(
93 rec.get("star", False), rec.get("tags", []), rec.get("note", "")
94 )
95 if not is_empty_entry(entry):
96 screen_id = rec.get("screen_id")
97 key = (
98 (str(pid), str(tid), str(screen_id))
99 if screen_id not in (None, "")
100 else (str(pid), str(tid))
101 )
102 store[key] = entry
103 return store
106def store_to_records(store: dict[Key, Entry]) -> list[dict]:
107 """Flatten a store into a sorted list of records for JSON export."""
108 records = []
109 for key, entry in sorted(store.items()):
110 pid, tid, *screen = key
111 record = {
112 "participant_id": pid,
113 "trial_id": tid,
114 "star": bool(entry.get("star", False)),
115 "tags": list(entry.get("tags", [])),
116 "note": str(entry.get("note", "")),
117 }
118 if screen:
119 record["screen_id"] = screen[0]
120 records.append(record)
121 return records
124def serialize(store: dict[Key, Entry], *, dataset: str | None = None) -> str:
125 """Serialize a store to a JSON document string.
127 ``dataset`` names the dataset the annotations were made on (DATA-48); the
128 file says so, and :func:`deserialize` does not need it back.
129 """
130 document: dict[str, object] = {"schema": SCHEMA_VERSION}
131 if dataset:
132 document["dataset"] = str(dataset)
133 document["annotations"] = store_to_records(store)
134 return json.dumps(document, indent=2)
137def file_dataset(text: str) -> str | None:
138 """The ``dataset`` an annotations file names, if any (schema 3+)."""
139 try:
140 data = json.loads(text)
141 except ValueError:
142 return None
143 name = data.get("dataset") if isinstance(data, dict) else None
144 return name if isinstance(name, str) and name else None
147class AnnotationsFileError(ValueError):
148 """A JSON document that is not an annotations file — wrong shape, not syntax."""
151_ID_TYPES = (str, int, float)
154def _check_record(index: int, record: object) -> None:
155 """Raise :class:`AnnotationsFileError` unless ``record`` has the exported shape."""
156 where = f"entry {index + 1}"
157 if not isinstance(record, dict):
158 raise AnnotationsFileError(f"{where} is not an annotation")
159 for field in ("participant_id", "trial_id"):
160 value = record.get(field)
161 if isinstance(value, bool) or not isinstance(value, _ID_TYPES):
162 raise AnnotationsFileError(f"{where} has no usable {field}")
163 screen_id = record.get("screen_id")
164 if screen_id is not None and (
165 isinstance(screen_id, bool) or not isinstance(screen_id, _ID_TYPES)
166 ):
167 raise AnnotationsFileError(f"{where} has an unusable screen_id")
168 if not isinstance(record.get("star", False), bool):
169 raise AnnotationsFileError(f"{where}: star must be true or false")
170 tags = record.get("tags", [])
171 if tags is not None and (
172 not isinstance(tags, list)
173 or any(isinstance(t, bool) or not isinstance(t, _ID_TYPES) for t in tags)
174 ):
175 raise AnnotationsFileError(f"{where}: tags must be a list of text labels")
176 note = record.get("note", "")
177 if note is not None and not isinstance(note, str):
178 raise AnnotationsFileError(f"{where}: note must be text")
181def deserialize(text: str) -> dict[Key, Entry]:
182 """Parse a JSON document (object with ``annotations`` or a bare list).
184 The shape is checked before anything is built, so a JSON file that is not
185 an annotations file — another app's, a settings file, a hand edit gone
186 wrong — raises :class:`AnnotationsFileError` (a ``ValueError``) rather than
187 importing nothing or failing half-way. Invalid JSON raises ``ValueError``
188 from :func:`json.loads`.
189 """
190 data = json.loads(text)
191 if isinstance(data, dict):
192 if "annotations" not in data:
193 raise AnnotationsFileError("it has no annotations list")
194 records = data["annotations"]
195 else:
196 records = data
197 if not isinstance(records, list):
198 raise AnnotationsFileError("its annotations are not a list")
199 for index, record in enumerate(records):
200 _check_record(index, record)
201 return records_to_store(records)
204def _trial_of(key: Key) -> tuple[str, str]:
205 return (key[0], key[1])
208def _trial_set(trials: Iterable[tuple[object, object]]) -> frozenset[tuple[str, str]]:
209 if isinstance(trials, frozenset):
210 return trials # already built by the caller, as strings
211 return frozenset((str(pid), str(tid)) for pid, tid in trials)
214def records_in(store: dict[Key, Entry], trials) -> list[dict]:
215 """The records of ``store`` on the trials in ``trials``, sorted.
217 ``trials`` is ``(participant_id, trial_id)`` pairs. Keeps what an Export
218 bundle writes to the trials it exports, and tells a dataset's annotations
219 on trials it still has from ones it no longer has (DATA-48). A screen
220 annotation belongs to its parent trial.
221 """
222 keep = _trial_set(trials)
223 return store_to_records(
224 {key: entry for key, entry in store.items() if _trial_of(key) in keep}
225 )
228def merge_records(
229 store: dict[Key, Entry], records: list[dict], trials
230) -> tuple[int, int]:
231 """Add ``records`` on the trials in ``trials`` to ``store``, in place.
233 An imported entry replaces the one already on its key — the file is the
234 newer statement about that trial — and a record for a trial the dataset
235 does not have is skipped, as a backup restore skips what does not match
236 (UX-174 r2). Returns ``(applied, skipped)``.
237 """
238 keep = _trial_set(trials)
239 incoming = records_to_store(records)
240 # A file saved before composite ids escaped a `_` inside a part names those
241 # trials by their old spelling; read it the dataset's way when that is
242 # unambiguous (`data.respell_reading`).
243 if any(_trial_of(key) not in keep for key in incoming):
244 incoming = {
245 (
246 key
247 if _trial_of(key) in keep
248 else (*respell_reading(key[0], key[1], keep), *key[2:])
249 ): entry
250 for key, entry in incoming.items()
251 }
252 applied = {key: entry for key, entry in incoming.items() if _trial_of(key) in keep}
253 store.update(applied)
254 return len(applied), len(incoming) - len(applied)
257def drop_records(store: dict[Key, Entry], records: list[dict]) -> int:
258 """Remove the entries ``records`` name from ``store``. Returns how many."""
259 removed = 0
260 for record in records:
261 pid, tid = str(record["participant_id"]), str(record["trial_id"])
262 screen_id = record.get("screen_id")
263 key = (pid, tid, str(screen_id)) if screen_id not in (None, "") else (pid, tid)
264 if store.pop(key, None) is not None:
265 removed += 1
266 return removed
269# ---------------------------------------------------------------------------
270# Session-backed layer
271# ---------------------------------------------------------------------------
274def _store() -> dict[Key, Entry]:
275 return st.session_state.setdefault(ANNOTATIONS_STATE_KEY, {})
278# ---------------------------------------------------------------------------
279# DATA-48 — the annotations belong to a dataset
280#
281# `ANNOTATIONS_STATE_KEY` holds the selected dataset's store; every other
282# dataset's waits in `DATASET_STORE_KEY`, and `activate_dataset` swaps them in
283# and out when the selection changes — DATA-47's design for the metadata
284# tables, taken over rather than reinvented, so the two follow one selection by
285# one set of rules. A dataset component in every key was the alternative; it
286# would have made each of the store's readers (the editor, the pickers'
287# markers, the filters, the Data page, the Export bundle) pass a dataset, where
288# the swap leaves them all reading the one store they already read.
289#
290# These functions take the session mapping explicitly, like `metadata`'s, so
291# the swap is testable on a plain dict.
292# ---------------------------------------------------------------------------
294#: Every dataset's annotations but the selected one's: ``{dataset: {key:
295#: entry}}``. The selected dataset is never in it.
296DATASET_STORE_KEY = "_annotations_by_dataset"
297#: Which dataset :data:`ANNOTATIONS_STATE_KEY`'s store belongs to right now.
298#: Absent until the first :func:`activate_dataset`.
299OWNER_KEY = "_annotations_owner"
300#: Entries that belong to no dataset yet — a pre-DATA-48 recovery cache's that
301#: no upload's trials claimed (:func:`restore_payload`). The first *real*
302#: dataset activated adopts them; the add wizard's never does, since nothing
303#: of it is cached and they would be lost with it.
304UNASSIGNED_KEY = "_annotations_unassigned"
305#: Bumped on every change to the stored (not live) stores — the recovery
306#: cache's cheap "did it change" test (:func:`store_signature`), the pattern
307#: of ``metadata.STORE_REVISION_KEY``: serializing every dataset's annotations
308#: on every rerun is not cheap once there are several thousand.
309STORE_REVISION_KEY = "_annotations_store_revision"
310#: The add-dataset wizard's dataset, before it has a name. The same token as
311#: ``metadata.PENDING_DATASET`` (pinned by a test; not imported, since
312#: `metadata` pulls in the data layer). Never cached.
313PENDING_DATASET = "\x00pending"
314#: CMP-8's key prefix for scanpath B's filters (``tabs._COMPARE_FILTER_PREFIX``)
315#: and the picker's "same dataset" answer (``compare_source.THIS_DATASET``).
316#: Pinned equal by a test; not imported, since both modules import this one.
317_COMPARE_PREFIX = "cmp"
318_COMPARE_SAME_DATASET = "This dataset"
321def _stored(session) -> dict[str, dict[Key, Entry]]:
322 store = session.get(DATASET_STORE_KEY)
323 return dict(store) if isinstance(store, dict) else {}
326def _live(session) -> dict[Key, Entry]:
327 live = session.get(ANNOTATIONS_STATE_KEY)
328 return dict(live) if isinstance(live, dict) else {}
331def _unassigned(session) -> dict[Key, Entry]:
332 pool = session.get(UNASSIGNED_KEY)
333 return dict(pool) if isinstance(pool, dict) else {}
336def _set_unassigned(session, pool: dict) -> None:
337 if pool:
338 session[UNASSIGNED_KEY] = pool
339 else:
340 session.pop(UNASSIGNED_KEY, None)
343def _bump(session) -> None:
344 session[STORE_REVISION_KEY] = int(session.get(STORE_REVISION_KEY) or 0) + 1
347def _reseed_editors_in(session) -> None:
348 """Drop the per-trial editors' widget state from ``session``.
350 Their keys carry the trial's ids, not the dataset's: left in place across a
351 swap, the editor of a trial the next dataset shares would seed from the last
352 dataset's values and write them into the new dataset's store. Nothing typed
353 is lost by it — every editor field saves itself on change (``_save_entry``),
354 and a widget's callback runs before the run that swaps.
355 """
356 for key in [
357 k
358 for k in list(session.keys())
359 if isinstance(k, str) and k.startswith(_WIDGET_PREFIX)
360 ]:
361 del session[key]
364def activate_dataset(session, dataset: str, *, adopt: bool = True) -> bool:
365 """Make ``dataset``'s annotations the session store; whether it changed.
367 Called by ``app.main`` on every run with the dataset on screen. When that
368 changed, the outgoing dataset's store is filed away and the incoming one's
369 comes back. A store with no owner yet (entries put there before the first
370 run) joins the :data:`UNASSIGNED_KEY` pool, which :func:`adopt_unassigned`
371 hands on. ``app.main`` passes ``adopt=False`` and adopts only
372 once the load has said which dataset is really shown (a missing corpus is
373 shown as the demo); the default adopts here, for callers with no load.
374 """
375 dataset = str(dataset)
376 owner = session.get(OWNER_KEY)
377 if owner == dataset:
378 if adopt:
379 adopt_unassigned(session)
380 return False
381 store = _stored(session)
382 live = _live(session)
383 unassigned = _unassigned(session)
384 if owner is None:
385 unassigned = {**live, **unassigned}
386 elif live:
387 store[str(owner)] = live
388 else:
389 store.pop(str(owner), None)
390 session[ANNOTATIONS_STATE_KEY] = store.pop(dataset, None) or {}
391 session[DATASET_STORE_KEY] = store
392 _set_unassigned(session, unassigned)
393 session[OWNER_KEY] = dataset
394 _bump(session)
395 _reseed_editors_in(session)
396 if adopt:
397 adopt_unassigned(session)
398 return True
401def adopt_unassigned(session) -> int:
402 """Give the :data:`UNASSIGNED_KEY` pool to the dataset on screen; how many.
404 The pool holds a pre-DATA-48 cache's entries that no restored upload's
405 trials claimed (:func:`restore_payload`) — so they are on trials of a
406 built-in or public corpus, and only such a dataset may adopt them. Never
407 an upload (each was asked in the claim pass, and said no), and never the
408 add wizard's unnamed dataset, which is not cached. The dataset's own entry
409 wins a collision with an adopted one.
410 """
411 pool = _unassigned(session)
412 owner = session.get(OWNER_KEY)
413 if not pool or owner is None or owner == PENDING_DATASET:
414 return 0
415 uploads = session.get("_datasets")
416 if isinstance(uploads, dict) and owner in uploads:
417 return 0
418 session[ANNOTATIONS_STATE_KEY] = {**pool, **_live(session)}
419 session.pop(UNASSIGNED_KEY, None)
420 _bump(session)
421 _reseed_editors_in(session)
422 return len(pool)
425def begin_pending_dataset(session) -> None:
426 """Start the add-dataset wizard's dataset with no annotations of its own."""
427 store = _stored(session)
428 if store.pop(PENDING_DATASET, None) is not None:
429 session[DATASET_STORE_KEY] = store
430 _bump(session)
433def adopt_pending_dataset(session, dataset: str) -> None:
434 """✅ Add dataset: the wizard's store becomes ``dataset``'s.
436 Only while the wizard's dataset is the selected one — relabelling another
437 dataset's store as the new one's would move its annotations — and the new
438 dataset starts clean: a stale store left under its name is dropped, not
439 merged (``wizard._safe_dataset_name`` keeps such names from being chosen).
440 """
441 begin_pending_dataset(session)
442 if session.get(OWNER_KEY) != PENDING_DATASET:
443 return
444 dataset = str(dataset)
445 store = _stored(session)
446 if store.pop(dataset, None) is not None:
447 session[DATASET_STORE_KEY] = store
448 session[OWNER_KEY] = dataset
449 _bump(session)
452def forget_dataset(session, dataset: str) -> None:
453 """A removed dataset's annotations go with it."""
454 dataset = str(dataset)
455 store = _stored(session)
456 if store.pop(dataset, None) is not None:
457 session[DATASET_STORE_KEY] = store
458 if session.get(OWNER_KEY) == dataset:
459 session[ANNOTATIONS_STATE_KEY] = {}
460 session.pop(OWNER_KEY, None)
461 _reseed_editors_in(session)
462 _bump(session)
465def current_dataset(session) -> str | None:
466 """The dataset the session store belongs to; ``None`` before the first run
467 and while the add wizard's unnamed dataset is open."""
468 owner = session.get(OWNER_KEY)
469 return None if owner in (None, PENDING_DATASET) else str(owner)
472def dataset_names(session) -> set[str]:
473 """Every dataset name that holds (or owns) an annotation store."""
474 names = {str(name) for name, entries in _stored(session).items() if entries}
475 owner = session.get(OWNER_KEY)
476 if owner is not None:
477 names.add(str(owner))
478 names.discard(PENDING_DATASET)
479 return names
482def rename_dataset(session, old: str, new: str) -> bool:
483 """A renamed dataset keeps its annotations; whether they moved.
485 Refuses — changing nothing — when ``new`` already holds a store of its own,
486 which the rename would otherwise overwrite. ``wizard.rename_dataset`` never
487 asks for such a name (``_safe_dataset_name`` avoids :func:`dataset_names`).
488 """
489 old, new = str(old), str(new)
490 if old == new:
491 return True
492 store = _stored(session)
493 if new in store or session.get(OWNER_KEY) == new:
494 return False
495 if old in store:
496 session[DATASET_STORE_KEY] = {
497 (new if key == old else key): value for key, value in store.items()
498 }
499 if session.get(OWNER_KEY) == old:
500 session[OWNER_KEY] = new
501 _bump(session)
502 return True
505def store_for(session, dataset: str) -> dict[Key, Entry]:
506 """``dataset``'s annotation store — the live one while it is selected."""
507 dataset = str(dataset)
508 if session.get(OWNER_KEY) == dataset:
509 return _live(session)
510 return dict(_stored(session).get(dataset) or {})
513def store_for_prefix(prefix: str = "") -> dict[Key, Entry]:
514 """The store the trial filters under key ``prefix`` narrow by (DATA-48).
516 The main pool's filters read the selected dataset's. Compare mode's
517 scanpath B (the ``cmp`` prefix) can come from another dataset, and then its
518 ⭐ / tag filters, its tag list and its picker's markers are *that*
519 dataset's — the rule ``metadata.attached_for`` follows for its tables.
520 Outside a script run (API, CLI) there is no store: ``{}``.
521 """
522 try:
523 session = st.session_state
524 if prefix != _COMPARE_PREFIX:
525 return _live(session)
526 other = session.get(COMPARE_SOURCE_STATE_KEY)
527 except Exception: # no script run context
528 return {}
529 if not other or other == _COMPARE_SAME_DATASET:
530 return _live(session)
531 return store_for(session, str(other))
534def upload_trials(entry) -> frozenset[tuple[str, str]]:
535 """An upload's ``(participant, trial)`` pairs, as annotations are keyed.
537 The trial picker's ids: `utils.build_combo_options` lists the fixation
538 table's trials under ``unique_trial_id`` when there is one, else
539 ``trial_id``. A table without fixations is read from its words instead.
540 """
541 for key in ("fixations", "words"):
542 frame = entry.get(key) if isinstance(entry, dict) else None
543 if not isinstance(frame, pd.DataFrame) or frame.empty:
544 continue
545 trial_col = "unique_trial_id" if "unique_trial_id" in frame else "trial_id"
546 if {"participant_id", trial_col} <= set(frame.columns):
547 pairs = frame[["participant_id", trial_col]].drop_duplicates()
548 return frozenset(
549 zip(pairs["participant_id"].astype(str), pairs[trial_col].astype(str))
550 )
551 return frozenset()
554def dataset_records(session) -> dict[str, list[dict]]:
555 """Every dataset's annotations, the selected one's live: ``{name: records}``.
557 What the recovery cache writes. The wizard's unnamed dataset is left out,
558 and so are entries no dataset owns yet — :func:`unassigned_records`.
559 """
560 stores = _stored(session)
561 owner = session.get(OWNER_KEY)
562 if owner is not None:
563 stores[str(owner)] = _live(session)
564 stores.pop(PENDING_DATASET, None)
565 return {
566 name: store_to_records(entries)
567 for name, entries in sorted(stores.items())
568 if entries
569 }
572def unassigned_records(session) -> list[dict]:
573 """Entries no dataset owns yet — cached as such, so none is lost.
575 The :data:`UNASSIGNED_KEY` pool, plus the session store itself while no
576 dataset owns it (a run that returned before :func:`activate_dataset`).
577 """
578 pool = _unassigned(session)
579 if session.get(OWNER_KEY) is None:
580 pool = {**_live(session), **pool}
581 return store_to_records(pool)
584def cache_payload(session) -> dict:
585 """The recovery cache's ``annotations`` value: ``{"datasets": …}`` (DATA-48)."""
586 payload: dict[str, object] = {"datasets": dataset_records(session)}
587 unassigned = unassigned_records(session)
588 if unassigned:
589 payload["unassigned"] = unassigned
590 return payload
593def store_signature(session) -> list:
594 """A cheap fingerprint of every dataset's annotations, for the cache.
596 The live store by content (the editor writes it directly), the rest by
597 :data:`STORE_REVISION_KEY` — ``metadata.store_signature``'s pattern — so a
598 rerun does not serialize every stored dataset's annotations.
599 """
600 return [
601 int(session.get(STORE_REVISION_KEY) or 0),
602 str(session.get(OWNER_KEY)),
603 store_to_records(_live(session)),
604 store_to_records(_unassigned(session)),
605 ]
608def _valid_records(records) -> list[dict]:
609 """One malformed record costs that record, not the rest."""
610 return [
611 record
612 for record in (records if isinstance(records, list) else [])
613 if isinstance(record, dict)
614 ]
617def restore_payload(session, payload) -> int:
618 """Put back what :func:`cache_payload` wrote; how many annotations landed.
620 A dataset this session already holds annotations for keeps its own — the
621 restore's ``setdefault`` rule.
623 **The one-time migration.** A manifest written before DATA-48 stored one
624 flat list of records naming no dataset; it is read as ``unassigned``. Each
625 entry goes, first, to every upload whose trials include it — the uploads'
626 frames are back in ``_datasets`` by now (``persistence.restore_state``
627 restores them first), and an annotation on a trial only one dataset has is
628 that dataset's. An entry no upload claims (one on a built-in or public
629 corpus, which is not in memory to ask) waits in :data:`UNASSIGNED_KEY` for
630 the first real dataset the session opens — the one the manifest had
631 selected, unless a link names another. Nothing is dropped, and the next save
632 writes everything back per dataset, so the conversion happens once.
633 """
634 if isinstance(payload, list):
635 datasets, unassigned = {}, payload
636 elif isinstance(payload, dict):
637 datasets = payload.get("datasets")
638 datasets = datasets if isinstance(datasets, dict) else {}
639 unassigned = payload.get("unassigned") or []
640 else:
641 datasets, unassigned = {}, []
642 live = _live(session)
643 owner = session.get(OWNER_KEY)
644 store = _stored(session)
645 restored = 0
647 def file_under(name: str, entries: dict) -> None:
648 nonlocal live
649 if name == owner:
650 live = {**entries, **live}
651 else:
652 store[name] = {**entries, **(store.get(name) or {})}
654 for name, records in datasets.items():
655 name = str(name)
656 entries = records_to_store(_valid_records(records))
657 if not entries or name == PENDING_DATASET or name in store:
658 continue
659 if name == owner and live:
660 continue
661 file_under(name, entries)
662 restored += len(entries)
664 legacy = records_to_store(_valid_records(unassigned))
665 if legacy:
666 claimed: set[Key] = set()
667 uploads = session.get("_datasets")
668 for name, entry in (uploads if isinstance(uploads, dict) else {}).items():
669 trials = upload_trials(entry)
670 mine = {k: e for k, e in legacy.items() if _trial_of(k) in trials}
671 if mine:
672 file_under(str(name), mine)
673 claimed.update(mine)
674 rest = {k: e for k, e in legacy.items() if k not in claimed}
675 _set_unassigned(session, {**rest, **_unassigned(session)})
676 restored += len(legacy)
677 session[ANNOTATIONS_STATE_KEY] = live
678 if store:
679 session[DATASET_STORE_KEY] = store
680 _bump(session)
681 return restored
684def payload_count(payload) -> int:
685 """How many annotations a cached ``annotations`` value holds, either shape."""
686 if isinstance(payload, list):
687 return len(payload)
688 if not isinstance(payload, dict):
689 return 0
690 datasets = payload.get("datasets")
691 total = sum(
692 len(records)
693 for records in (datasets.values() if isinstance(datasets, dict) else [])
694 if isinstance(records, list)
695 )
696 unassigned = payload.get("unassigned")
697 return total + (len(unassigned) if isinstance(unassigned, list) else 0)
700def get_entry(
701 participant_id: str, trial_id: str, screen_id: str | None = None
702) -> Entry:
703 key = (
704 (str(participant_id), str(trial_id), str(screen_id))
705 if screen_id not in (None, "")
706 else (str(participant_id), str(trial_id))
707 )
708 return _store().get(key, default_entry())
711def set_entry(
712 participant_id: str,
713 trial_id: str,
714 *,
715 star: bool,
716 tags: list[str],
717 note: str,
718 screen_id: str | None = None,
719) -> None:
720 """Upsert an annotation; empty entries are pruned to keep the store small."""
721 key = (
722 (str(participant_id), str(trial_id), str(screen_id))
723 if screen_id not in (None, "")
724 else (str(participant_id), str(trial_id))
725 )
726 entry = _normalize_entry(star, tags, note)
727 store = _store()
728 if is_empty_entry(entry):
729 store.pop(key, None)
730 else:
731 store[key] = entry
734def known_tags(prefix: str = "", *, trial_level: bool = False) -> list[str]:
735 """Preset tags plus any tag used in the dataset's store, sorted.
737 ``prefix`` picks the dataset as :func:`store_for_prefix` does, so compare
738 mode's scanpath B lists its own dataset's tags (DATA-48).
740 ``trial_level`` leaves out tags used only on screen annotations — what the
741 trial filters offer, since they read the trial's own entry (:func:`select_keys`)
742 and a screen-only tag there could never match.
743 """
744 tags: set[str] = set(PRESET_TAGS)
745 store = store_for_prefix(prefix) if prefix else _store()
746 for key, entry in store.items():
747 if trial_level and len(key) > 2:
748 continue
749 tags.update(entry.get("tags", []))
750 return sorted(tags)
753def has_screen_annotations(prefix: str = "") -> bool:
754 """Whether the dataset's store holds any screen annotation."""
755 store = store_for_prefix(prefix) if prefix else _store()
756 return any(len(key) > 2 for key in store)
759def current_records() -> list[dict]:
760 """The open dataset's annotations as records — the Export bundle's (UX-179)."""
761 return store_to_records(_store())
764def _reseed_trial_editors() -> None:
765 """Drop the per-trial editors' widget state, so they re-seed from the store."""
766 _reseed_editors_in(st.session_state)
769# ---------------------------------------------------------------------------
770# UI render helpers
771# ---------------------------------------------------------------------------
774def _add_tag_callback(tags_key: str, newtag_key: str, save_args: tuple) -> None:
775 """on_change for the 'add tag' input: append to the multiselect's state.
777 Runs before the next rerun, so writing the multiselect's session_state here
778 is allowed (the widget hasn't been instantiated yet that run). Saves the
779 entry too, like every other editor field (see :func:`_save_entry_callback`)."""
780 new_tag = str(st.session_state.get(newtag_key, "")).strip()
781 if not new_tag:
782 return
783 current = list(st.session_state.get(tags_key, []))
784 if new_tag not in current:
785 st.session_state[tags_key] = current + [new_tag]
786 st.session_state[newtag_key] = ""
787 _save_entry_callback(*save_args)
790def _save_entry_callback(
791 participant_id: str,
792 trial_id: str,
793 screen_id: str | None,
794 star_key: str,
795 tags_key: str,
796 note_key: str,
797) -> None:
798 """Persist the editor's fields as soon as one changes, before the rerun.
800 Widget callbacks run before the app body. Without this the picker reads the
801 previous annotation store, then ``render_trial_annotations`` saves the new
802 value later in the run — making the star marker appear one click behind the
803 checkbox. And (DATA-48) the run that switches dataset drops the editors'
804 widget state in ``activate_dataset`` before any editor renders: a tag or a
805 note saved only by the render body would be lost with it. Saved here, it is
806 already in the outgoing dataset's store when the swap files that away.
807 """
808 entry = get_entry(participant_id, trial_id, screen_id)
809 state = st.session_state
810 set_entry(
811 participant_id,
812 trial_id,
813 star=bool(state[star_key]) if star_key in state else bool(entry["star"]),
814 tags=list(state[tags_key]) if tags_key in state else list(entry["tags"]),
815 note=str(state[note_key]) if note_key in state else str(entry["note"]),
816 screen_id=screen_id,
817 )
820def widget_slug(
821 participant_id: str, trial_id: str, screen_id: str | None = None
822) -> str:
823 """The per-trial editor's widget-key suffix: the store key, unambiguously.
825 BUG-110: joining the ids with ``__`` and writing the parent scope as the
826 word ``parent`` gave ``("a__b", "c")`` and ``("a", "b__c")`` — or a trial's
827 parent and its screen named ``parent`` — the same widget keys, so opening
828 the one seeded its editor from the other's values and saved them as its
829 own. A JSON list keeps every id whole and writes the parent as ``null``.
830 """
831 ident = [str(participant_id), str(trial_id)]
832 ident.append(None if screen_id is None else str(screen_id))
833 return json.dumps(ident, ensure_ascii=False)
836def render_trial_annotations(
837 participant_id: str,
838 trial_id: str,
839 *,
840 screen_id: str | None = None,
841 bare: bool = False,
842) -> None:
843 """Render the per-trial annotations (star / tags / notes).
845 ``bare=True`` drops the expander wrapper so it can sit inside a subtab."""
846 annotation_screen = None
847 if screen_id is not None:
848 scope_key = f"{_WIDGET_PREFIX}scope_{widget_slug(participant_id, trial_id)}"
849 scope = panel_field(
850 st,
851 "radio",
852 "Annotation scope",
853 options=["Parent trial", "This screen"],
854 # #374: the stored option keeps its name; "parent" is internal.
855 format_func=lambda option: (
856 "Whole trial" if option == "Parent trial" else option
857 ),
858 key=scope_key,
859 horizontal=True,
860 help="Whole trial: every screen of this trial. This screen: only the "
861 "screen on view.",
862 )
863 if scope == "This screen":
864 annotation_screen = str(screen_id)
865 entry = get_entry(participant_id, trial_id, annotation_screen)
866 slug = widget_slug(participant_id, trial_id, annotation_screen)
867 star_key = f"{_WIDGET_PREFIX}star_{slug}"
868 tags_key = f"{_WIDGET_PREFIX}tags_{slug}"
869 note_key = f"{_WIDGET_PREFIX}note_{slug}"
870 newtag_key = f"{_WIDGET_PREFIX}newtag_{slug}"
871 save_args = (
872 participant_id,
873 trial_id,
874 annotation_screen,
875 star_key,
876 tags_key,
877 note_key,
878 )
880 # Seed widget state once from the store (re-seeds after a JSON import, which
881 # clears these keys).
882 st.session_state.setdefault(star_key, entry["star"])
883 st.session_state.setdefault(tags_key, list(entry["tags"]))
884 st.session_state.setdefault(note_key, entry["note"])
886 label = f"{ICONS['annotations']} Annotations" + (
887 f" {ICONS['favorite']}" if entry["star"] else ""
888 )
889 if entry["tags"]:
890 label += f" · {', '.join(entry['tags'])}"
891 container = st.container() if bare else st.expander(label, expanded=False)
892 with container:
893 # UX-69: `label | field` rows, so the five stacked controls fit under the
894 # plot without scrolling. Unlike the rail (`controls._labeled`'s note),
895 # the checkbox is split too: here the label column is the row spine every
896 # other field lines up on, and a native checkbox would put its box —
897 # not its title — at that edge.
898 star = panel_field(
899 st,
900 "checkbox",
901 f"{ICONS['favorite']} Favorite (star this trial)",
902 display=f"{ICONS['favorite']} Favorite",
903 key=star_key,
904 help="Star this trial (with scope *This screen*, this screen only).",
905 on_change=_save_entry_callback,
906 args=save_args,
907 )
908 # Options must include every currently-selected tag (incl. ones added
909 # via the input) or st.multiselect raises.
910 options = sorted(
911 set(known_tags())
912 | set(entry["tags"])
913 | set(st.session_state.get(tags_key, []))
914 )
915 tags_help = "Select tags or add one."
916 label_col, tags_col, add_col = st.columns(
917 [PANEL_LABEL_W, 0.52, 0.28],
918 gap="xsmall",
919 vertical_alignment="center",
920 )
921 row_label(label_col, "Tags", tags_help)
922 # ENG-49: built directly in a column, so it does not go through
923 # `fields.labeled` and needs its own `wrap` — a trial's tags are the
924 # thing this row exists to show, and 1.63 would otherwise scroll all
925 # but the first one or two out of sight.
926 tags = tags_col.multiselect(
927 "Tags",
928 options=options,
929 key=tags_key,
930 help=tags_help,
931 label_visibility="collapsed",
932 wrap=True,
933 on_change=_save_entry_callback,
934 args=save_args,
935 )
936 add_col.text_input(
937 "Add a tag",
938 key=newtag_key,
939 placeholder="Add a new tag",
940 on_change=_add_tag_callback,
941 args=(tags_key, newtag_key, save_args),
942 label_visibility="collapsed",
943 )
944 # Top-aligned: the box is ~100px tall, and a centred title would float
945 # opposite the middle of an empty note rather than beside its first line.
946 note = panel_field(
947 st,
948 "text_area",
949 "Notes",
950 align="top",
951 key=note_key,
952 placeholder="Researcher notes for this trial…",
953 height=100,
954 on_change=_save_entry_callback,
955 args=save_args,
956 )
957 set_entry(
958 participant_id,
959 trial_id,
960 star=star,
961 tags=tags,
962 note=note,
963 screen_id=annotation_screen,
964 )
965 # UX-76: no shortcut button under this — the caption names where the
966 # whole dataset's annotations are listed instead.
967 st.caption(
968 "Every annotation on this dataset is listed on "
969 f"{ICONS['view_data']} **Data Management → Annotations**, to export, import "
970 "or delete."
971 )
974def select_keys(
975 store: dict[Key, Entry],
976 keys: list[Key],
977 *,
978 favorites_only: bool = False,
979 required_tags: list[str] | None = None,
980 excluded_tags: list[str] | None = None,
981) -> list[Key]:
982 """Pure core of :func:`filter_keys` — filter ``keys`` against ``store``.
984 Trial level only: each key is looked up as given, so a parent
985 ``(participant_id, trial_id)`` key reads the trial's own annotation and
986 never a screen's. That is the filters' stated scope — a screen star or tag
987 neither keeps nor drops its trial, and
988 the panel offers only trial-level tags (:func:`known_tags`).
990 - ``favorites_only``: keep only starred trials.
991 - ``required_tags``: keep trials carrying *any* of these tags.
992 - ``excluded_tags``: drop trials carrying *any* of these tags.
993 """
994 required = set(required_tags or [])
995 excluded = set(excluded_tags or [])
996 out: list[Key] = []
997 for key in keys:
998 entry = store.get(key)
999 tags = set(entry.get("tags", [])) if entry else set()
1000 starred = bool(entry.get("star")) if entry else False
1001 if favorites_only and not starred:
1002 continue
1003 if required and not (tags & required):
1004 continue
1005 if excluded and (tags & excluded):
1006 continue
1007 out.append(key)
1008 return out
1011def filter_keys(
1012 keys: list[Key],
1013 *,
1014 favorites_only: bool = False,
1015 required_tags: list[str] | None = None,
1016 excluded_tags: list[str] | None = None,
1017 prefix: str = "",
1018) -> list[Key]:
1019 """Session-backed wrapper around :func:`select_keys`.
1021 ``prefix`` picks the dataset's store as :func:`store_for_prefix` does.
1022 """
1023 return select_keys(
1024 store_for_prefix(prefix) if prefix else _store(),
1025 keys,
1026 favorites_only=favorites_only,
1027 required_tags=required_tags,
1028 excluded_tags=excluded_tags,
1029 )
1032# ---------------------------------------------------------------------------
1033# 🗂️ Data → Annotations (UX-174 r2)
1034# ---------------------------------------------------------------------------
1036#: The dataset tab's widget keys. The nonce is bumped after an import or a
1037#: delete, which gives the uploader and the table fresh keys: an uploader keeps
1038#: its file (and would import it again on every rerun), and a table's row
1039#: selection is by position, so it would point at other rows once some are gone.
1040_DATASET_NONCE_KEY = "_dataset_annotations_nonce"
1041_DATASET_NOTE_KEY = "_dataset_annotations_note"
1044def _dataset_widget_key(name: str) -> str:
1045 return f"dataset_annotations_{name}_{st.session_state.get(_DATASET_NONCE_KEY, 0)}"
1048def _refresh_dataset_widgets(note: str) -> None:
1049 st.session_state[_DATASET_NONCE_KEY] = (
1050 int(st.session_state.get(_DATASET_NONCE_KEY, 0)) + 1
1051 )
1052 st.session_state[_DATASET_NOTE_KEY] = note
1053 _reseed_trial_editors()
1056def _plural(count: int, noun: str) -> str:
1057 return f"{count:,} {noun}{'' if count == 1 else 's'}"
1060def _import_dataset_annotations(
1061 uploader_key: str, trials: frozenset, dataset_name: str = ""
1062) -> None:
1063 upload = st.session_state.get(uploader_key)
1064 if upload is None:
1065 return
1066 try:
1067 text = upload.getvalue().decode("utf-8")
1068 records = store_to_records(deserialize(text))
1069 except (UnicodeDecodeError, ValueError) as exc:
1070 # Nothing was merged yet, so the dataset's annotations are untouched,
1071 # and the fresh uploader key below lets another file be chosen.
1072 reason = f" ({exc})" if isinstance(exc, AnnotationsFileError) else ""
1073 st.session_state[_DATASET_NOTE_KEY] = (
1074 f"error:That file is not an annotations JSON file{reason}. "
1075 "Nothing was imported."
1076 )
1077 st.session_state[_DATASET_NONCE_KEY] = (
1078 int(st.session_state.get(_DATASET_NONCE_KEY, 0)) + 1
1079 )
1080 return
1081 # DATA-48: into the open dataset, whatever the file names — a file from
1082 # before annotations were per dataset names none, and one exported from
1083 # another dataset is still the user's call to bring here. The name is said.
1084 applied, skipped = merge_records(_store(), records, trials)
1085 note = f"Imported {_plural(applied, 'annotation')}."
1086 source = file_dataset(text)
1087 if source and source != dataset_name:
1088 note = f"Imported {_plural(applied, 'annotation')} exported from **{source}**."
1089 if skipped:
1090 note += (
1091 f" Skipped {_plural(skipped, 'annotation')} on trials this dataset "
1092 "doesn't have."
1093 )
1094 _refresh_dataset_widgets(note)
1097#: Cell text of a row's **Open** button — `ButtonColumn` takes its label from
1098#: the cell value, as `tabs._OPEN_TRIAL_LABEL` does.
1099_OPEN_LABEL = f"{ICONS['open']} Open"
1102def _open_annotation(
1103 click_key: str, records: list[dict], trials: frozenset, open_trials: frozenset
1104) -> None:
1105 """A row's **Open**: show its reading (and screen) in the Scanpath view.
1107 The click is a callback, so it parks the request with
1108 ``url_state.request_trial`` — the Corpus Analysis tables' hop. It opens
1109 only that exact reading: one the dataset hasn't loaded, or one the trial
1110 filters hide, is explained here instead, since the picker would otherwise
1111 land on another reader's trial of the same id or stay where it was.
1112 """
1113 click = st.session_state.get(click_key)
1114 row = click.get("row") if isinstance(click, dict) else None
1115 if row is None or not 0 <= row < len(records):
1116 return
1117 record = records[row]
1118 pid, tid = str(record["participant_id"]), str(record["trial_id"])
1119 where = f"participant **{pid}**, trial **{tid}**"
1120 if (pid, tid) not in trials:
1121 st.session_state[_DATASET_NOTE_KEY] = (
1122 f"error:Can't open {where}: this dataset hasn't loaded that trial."
1123 )
1124 return
1125 if (pid, tid) not in open_trials:
1126 st.session_state[_DATASET_NOTE_KEY] = (
1127 f"error:Can't open {where}: the trial filters hide it. Clear or "
1128 f"change the filters on {ICONS['view_scanpath']} **Scanpath**, "
1129 "then open it again."
1130 )
1131 return
1132 # Imported at call time: `url_state` imports the controls, which import
1133 # this module.
1134 from .url_state import request_trial
1136 request_trial(pid, tid, screen_id=record.get("screen_id"))
1139def _delete_dataset_annotations(records: list[dict]) -> None:
1140 removed = drop_records(_store(), records)
1141 _refresh_dataset_widgets(f"Deleted {_plural(removed, 'annotation')}.")
1144def _annotations_frame(
1145 records: list[dict], trials: frozenset, trial_labels=None
1146) -> pd.DataFrame:
1147 """The Annotations table. ``trial_labels`` maps a trial id to the trial
1148 picker's label for it (#374 F5: one trial label everywhere)."""
1149 shown = trial_labels or {}
1150 frame = pd.DataFrame(
1151 {
1152 "Participant": [r["participant_id"] for r in records],
1153 "Trial": [
1154 shown.get(str(r["trial_id"]), str(r["trial_id"])) for r in records
1155 ],
1156 "Screen": [r.get("screen_id", "") for r in records],
1157 "Favorite": [r["star"] for r in records],
1158 "Tags": [r["tags"] for r in records],
1159 "Note": [r["note"] for r in records],
1160 "In dataset": [
1161 (str(r["participant_id"]), str(r["trial_id"])) in trials
1162 for r in records
1163 ],
1164 }
1165 )
1166 if not frame["Screen"].astype(bool).any():
1167 frame = frame.drop(columns="Screen")
1168 if frame["In dataset"].all():
1169 frame = frame.drop(columns="In dataset")
1170 return frame
1173def render_dataset_annotations(
1174 trials, *, dataset_name: str, open_trials=None, trial_labels=None
1175) -> None:
1176 """🗂️ Data → **Annotations**: every annotation the open dataset holds.
1178 One table — participant, trial, favorite, tags, note — with **Export** (this
1179 dataset's annotations as JSON), **Import** (the same file, or the
1180 ``annotations.json`` of an Export bundle: entries on trials the dataset has
1181 are added, the rest are skipped)
1182 and **Delete**, for the rows ticked in the table. Annotations are still made
1183 per trial, on 🗺️ Scanpath → Annotations; this is where they are seen whole.
1185 DATA-48: the session store *is* the dataset's now, so the table lists all
1186 of it — including any entry on a trial the dataset has not loaded: one made
1187 under another selection of a corpus' parts, or one a recovery cache from
1188 before annotations were per dataset assigned here (:func:`restore_payload`).
1189 Those are flagged rather than hidden, so an entry is never out of reach:
1190 exported from here, it imports into the dataset it belongs to.
1192 ``open_trials`` — the trials the Scanpath picker can show, after the trial
1193 filters — gives each row an **Open** button (:func:`_open_annotation`);
1194 without it the table has none. ``trial_labels`` writes each trial as
1195 the trial picker does.
1196 """
1197 trials = _trial_set(trials)
1198 records = store_to_records(_store())
1199 note = st.session_state.pop(_DATASET_NOTE_KEY, None)
1200 if note and note.startswith("error:"):
1201 st.error(note.removeprefix("error:"), icon=ICONS["error"])
1202 elif note:
1203 st.success(note, icon=ICONS["confirm"])
1205 toolbar = st.container(
1206 key="dataset_annotations_toolbar",
1207 horizontal=True,
1208 vertical_alignment="center",
1209 gap="small",
1210 )
1211 toolbar.download_button(
1212 "Export",
1213 icon=ICONS["download"],
1214 data=serialize(records_to_store(records), dataset=dataset_name),
1215 file_name=f"{_file_slug(dataset_name)}_annotations.json",
1216 mime="application/json",
1217 key="dataset_annotations_export",
1218 disabled=not records,
1219 help="Download this dataset's annotations as a JSON file.",
1220 )
1221 with toolbar.popover("Import", icon=ICONS["upload"]):
1222 uploader_key = _dataset_widget_key("import")
1223 st.file_uploader(
1224 "Annotations file (JSON)",
1225 type=["json"],
1226 key=uploader_key,
1227 on_change=_import_dataset_annotations,
1228 args=(uploader_key, trials, dataset_name),
1229 max_upload_size=upload_limit_mb(),
1230 )
1231 st.caption(
1232 "A file exported here or in an Export bundle. Annotations on trials "
1233 "this dataset doesn't have are skipped, and an imported one replaces "
1234 "what its trial already had."
1235 )
1236 delete_slot = toolbar.container(width="content")
1238 if not records:
1239 st.caption(
1240 "No annotations on this dataset yet. Star, tag or note a trial on "
1241 f"{ICONS['view_scanpath']} **Scanpath → Annotations**."
1242 )
1243 return
1244 elsewhere = len(records) - len(records_in(_store(), trials))
1245 if elsewhere:
1246 st.caption(
1247 f"{_plural(elsewhere, 'annotation')} here "
1248 f"{'is' if elsewhere == 1 else 'are'} on trials this dataset hasn't "
1249 "loaded (*In dataset* unticked). They stay with this "
1250 "dataset; to move them to another, **Export** them here and "
1251 "**Import** them there."
1252 )
1253 frame = _annotations_frame(records, trials, trial_labels)
1254 column_config = {}
1255 if open_trials is not None:
1256 frame.insert(0, "Open", _OPEN_LABEL)
1257 open_key = _dataset_widget_key("open")
1258 column_config["Open"] = st.column_config.ButtonColumn(
1259 "",
1260 type="tertiary",
1261 width="small",
1262 help="Show this trial — and its screen, for a screen annotation — "
1263 "in the Scanpath view.",
1264 on_click=_open_annotation,
1265 args=(open_key, records, trials, _trial_set(open_trials)),
1266 key=open_key,
1267 )
1268 event = st.dataframe(
1269 frame,
1270 hide_index=True,
1271 width="stretch",
1272 on_select="rerun",
1273 selection_mode="multi-row",
1274 key=_dataset_widget_key("table"),
1275 column_config={
1276 **column_config,
1277 "Favorite": st.column_config.CheckboxColumn("Favorite", width="small"),
1278 "Tags": st.column_config.ListColumn("Tags"),
1279 "Note": st.column_config.TextColumn("Note", width="large"),
1280 "In dataset": st.column_config.CheckboxColumn(
1281 "In dataset",
1282 width="small",
1283 help="Whether this dataset has the annotated trial.",
1284 ),
1285 },
1286 )
1287 picked = [records[i] for i in event.selection.rows if i < len(records)]
1288 with delete_slot.popover(
1289 f"Delete ({len(picked)})" if picked else "Delete",
1290 icon=ICONS["delete"],
1291 disabled=not picked,
1292 help="Tick rows in the table to delete them.",
1293 ):
1294 st.write(
1295 f"Delete {_plural(len(picked), 'annotation')}? This cannot be undone "
1296 "— **Export** first to keep a copy."
1297 )
1298 st.button(
1299 "Delete",
1300 icon=ICONS["delete"],
1301 key="dataset_annotations_delete",
1302 type="primary",
1303 on_click=_delete_dataset_annotations,
1304 args=(picked,),
1305 )
1308def _file_slug(name: str) -> str:
1309 slug = "".join(ch if ch.isalnum() else "_" for ch in str(name)).strip("_")
1310 return slug.lower() or "dataset"