Coverage for scanpath_studio/metadata.py: 93%
826 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Keyed, entity-level metadata tables (DATA-20).
3Milestone 1 is **participant grain**: a separate table whose rows are readers
4and whose columns (``native_language``, ``age``, ``comprehension_score``, …)
5should behave "as if they were fields in the data" — filterable, chip-able,
6sortable, inspectable, exportable — without being any part of the recorded eye
7movements.
9Two rules shape everything here.
11**The table stays separate.** It is never broadcast across every word/fixation
12row: a 40-row participant table joined onto three million fixations costs
13memory and buys nothing, and it would make a *reader* attribute look like a
14per-fixation measurement. Instead the frame is kept as-is and consumed at three
15narrow boundaries:
17* *filtering* — a participant-grain constraint is a **participant** constraint,
18 so :func:`participants_matching` turns a selection into the set of ids the
19 existing participant filter already knows how to apply. No join at all.
20* *projection* — :func:`project` left-joins chosen columns onto a **small**
21 frame (the per-trial ``combos`` table, a group-by result), which is where
22 sorting, grouping and chips read from.
23* *display / export* — the frame itself, shown and written as its own table.
25**Nothing is silently collapsed.** Duplicate participant rows that disagree are
26not resolved by taking the first one: the id is reported as *conflicting* and
27contributes no value, so a downstream field reads as missing rather than as an
28arbitrary winner. Unmatched ids are reported on both sides — rows describing
29readers who are not in the data, and readers in the data with no row.
31**Milestone 2 is trial grain (DATA-29)**: the same idea one level down — a
32table whose rows are *readings*. It reuses everything that is about validating a
33keyed table (the dtype classification, the field registry, the join report, the
34"conflicting rows are dropped, not resolved" rule) and differs only where the
35grain genuinely differs:
37* the **key** is the user's call — a trial id alone, or a reader **and** a trial
38 id, because a repeated reading is a different trial for the same reader only
39 in corpora that record it that way;
40* **filtering** narrows to a set of ``(participant_id, trial_id)`` keys, applied
41 by ``data.filter_to_keys`` — there is no participant-constraint indirection to
42 mirror, because a trial constraint already *is* the grain the pool is keyed on.
44Later grains (stimulus, screen, word, fixation) add rows to the same registry;
45:class:`MetadataField` already carries ``grain``.
46"""
48from __future__ import annotations
50import hashlib
51from collections.abc import Iterable, Mapping, Sequence
52from dataclasses import dataclass, replace
54import numpy as np
55import pandas as pd
57from . import data as _data
58from .data import (
59 composite_respelling_map,
60 stable_id,
61 trial_id_series,
62 trial_mapping_columns,
63 zero_padding_map,
64)
65from .session_keys import COMPARE_SOURCE_STATE_KEY
68def _with_data_candidates(data_candidates: list[str], *extras: str) -> tuple[str, ...]:
69 """``data``'s own candidate list, in its order, then the metadata-only
70 spellings it does not already hold (compared case-insensitively, as the
71 ``infer_*_id_column`` lookups compare).
73 DATA-44: the metadata lists used to be hand-copied twins of ``data``'s and
74 drifted — the text list lost ``unique_paragraph_id``, the demo corpus's own
75 text id — so a table exported beside the data needed a manual pick. Deriving
76 them means a name ``data`` learns is a name a metadata table is keyed by."""
77 seen: set[str] = set()
78 out: list[str] = []
79 for name in (*data_candidates, *extras):
80 if name.lower() not in seen:
81 seen.add(name.lower())
82 out.append(name)
83 return tuple(out)
86# Source columns that plausibly hold the reader id, most explicit first: the
87# names `data.PARTICIPANT_CANDIDATES` maps the reader from, so a metadata file
88# exported beside the data usually needs no picking, plus a few spellings only a
89# hand-made readers table tends to use. First hit wins, and the user can always
90# override the guess in the UI.
91PARTICIPANT_ID_CANDIDATES: tuple[str, ...] = _with_data_candidates(
92 _data.PARTICIPANT_CANDIDATES, "participant", "subject", "reader", "pid"
93)
95# Grain of a field — the entity one row describes. PARTICIPANT (DATA-20),
96# TRIAL (DATA-29) and TEXT are ingested; the rest are named so the registry's
97# shape is settled.
98GRAIN_PARTICIPANT = "participant"
99GRAIN_TRIAL = "trial"
100GRAIN_TEXT = "text"
102# Source columns that plausibly hold the trial id — the trial-grain twin of
103# PARTICIPANT_ID_CANDIDATES, derived from `data.TRIAL_CANDIDATES` the same way.
104TRIAL_ID_CANDIDATES: tuple[str, ...] = _with_data_candidates(
105 _data.TRIAL_CANDIDATES, "item_id"
106)
108# The text-grain twin, derived from `data.TEXT_ID_CANDIDATES`.
109TEXT_ID_CANDIDATES: tuple[str, ...] = _with_data_candidates(
110 _data.TEXT_ID_CANDIDATES, "text", "item_id", "stimulus_id"
111)
113# Loader bookkeeping, never user metadata: `data.read_tables` tags each row with
114# the file it came from, which would otherwise be registered as a field called
115# "Source file" and offered as a filter and a chip. Excluded here, in the one
116# place every ingestion route passes through, rather than at each caller.
117_BOOKKEEPING_COLUMNS = frozenset({"source_file"})
119# Session state: the validated table, and the raw frame it was built from (kept
120# so a different id column can be picked without re-uploading the file). Both
121# are plain session state rather than widget keys — they are not wire format,
122# and `session_keys.py` deliberately does not pin them.
123SESSION_KEY = "_participant_metadata"
124RAW_SESSION_KEY = "_participant_metadata_raw"
125FILE_SESSION_KEY = "_participant_metadata_file"
127# DATA-29 — the same three, for the trial table. Separate keys rather than one
128# keyed-by-grain dict: the two tables are attached, replaced and cleared
129# independently, and every consumer wants one of them specifically.
130TRIAL_SESSION_KEY = "_trial_metadata"
131TRIAL_RAW_SESSION_KEY = "_trial_metadata_raw"
132TRIAL_FILE_SESSION_KEY = "_trial_metadata_file"
134# The same three, for the text table (one row per text_id) — the third grain.
135TEXT_SESSION_KEY = "_text_metadata"
136TEXT_RAW_SESSION_KEY = "_text_metadata_raw"
137TEXT_FILE_SESSION_KEY = "_text_metadata_file"
139_DTYPE_CATEGORICAL = "categorical"
140_DTYPE_NUMERIC = "numeric"
141_DTYPE_BOOLEAN = "boolean"
144@dataclass(frozen=True)
145class MetadataField:
146 """One registered column, with everything a consumer needs to place it."""
148 name: str
149 label: str
150 grain: str
151 dtype: str
152 source: str
153 n_unique: int
154 n_missing: int
156 @property
157 def is_numeric(self) -> bool:
158 return self.dtype == _DTYPE_NUMERIC
160 @property
161 def is_categorical(self) -> bool:
162 return self.dtype in (_DTYPE_CATEGORICAL, _DTYPE_BOOLEAN)
165@dataclass(frozen=True)
166class JoinReport:
167 """What happened when the table met the participants actually loaded.
169 Every count is a list of ids rather than a number so the UI can name them —
170 "3 unmatched" is not actionable, "``p07``, ``p12``, ``p31``" is.
171 """
173 matched: tuple[str, ...] = ()
174 only_in_table: tuple[str, ...] = ()
175 only_in_data: tuple[str, ...] = ()
176 duplicated: tuple[str, ...] = ()
177 conflicting: tuple[str, ...] = ()
178 #: How many rows of the file were folded together because they repeated a
179 #: key without disagreeing (each field takes the one value its rows hold).
180 combined_rows: int = 0
182 @property
183 def is_clean(self) -> bool:
184 return not (
185 self.only_in_table
186 or self.only_in_data
187 or self.duplicated
188 or self.conflicting
189 )
192@dataclass(frozen=True)
193class ParticipantMetadata:
194 """A validated participant table plus its field registry.
196 ``frame`` is indexed by nothing in particular but always carries a string
197 ``participant_id`` column; conflicting ids have been dropped from it (and
198 named in :attr:`report`), so a lookup either finds one unambiguous row or
199 finds none.
200 """
202 frame: pd.DataFrame
203 fields: tuple[MetadataField, ...]
204 source_name: str
205 id_column: str
206 report: JoinReport = JoinReport()
208 @property
209 def names(self) -> tuple[str, ...]:
210 return tuple(field.name for field in self.fields)
212 def field(self, name: str) -> MetadataField | None:
213 for candidate in self.fields:
214 if candidate.name == name:
215 return candidate
216 return None
218 def values_for(self, participant_id) -> dict[str, object]:
219 """Every registered value for one reader (missing ids give ``{}``)."""
220 if self.frame.empty:
221 return {}
222 match = self.frame[self.frame["participant_id"] == str(participant_id)]
223 if match.empty:
224 return {}
225 row = match.iloc[0]
226 return {name: row[name] for name in self.names if name in match.columns}
228 @property
229 def joined_frame(self) -> pd.DataFrame:
230 """Only the rows describing readers that are actually loaded.
232 What the *controls* must be built from. Offering a value that belongs to
233 a reader the report has just called "not loaded — ignored" gives the
234 user a filter that can only ever empty the pool, and stretches a numeric
235 slider to a bound nobody in the data has. With no participant list to
236 join against (``participants=None``), the report matches everything and
237 this is the whole frame.
238 """
239 if self.frame.empty:
240 return self.frame
241 return self.frame[self.frame["participant_id"].isin(set(self.report.matched))]
243 def series(self, name: str) -> pd.Series:
244 """``participant_id`` → value for one field, for projection/lookup."""
245 if self.frame.empty or name not in self.frame.columns:
246 return pd.Series(dtype="object")
247 return self.frame.set_index("participant_id")[name]
250@dataclass(frozen=True)
251class TrialMetadata:
252 """A validated trial table plus its field registry (DATA-29).
254 ``frame`` always carries a string ``trial_id`` column, and a string
255 ``participant_id`` column as well when the table is keyed by both. Rows
256 whose key repeats *with different values* have been dropped and named in
257 :attr:`report`, so a lookup either finds one unambiguous row or none — the
258 same rule the participant table follows.
260 ``keyed_by_participant`` is the user's answer to the question DATA-29 opened
261 with. It is not inferred: a corpus where every reader reads every text can
262 key by trial id alone and mean it, and one with repeated readings cannot,
263 and nothing in the file itself says which world you are in.
264 """
266 frame: pd.DataFrame
267 fields: tuple[MetadataField, ...]
268 source_name: str
269 trial_column: str
270 participant_column: str | None = None
271 report: JoinReport = JoinReport()
273 @property
274 def keyed_by_participant(self) -> bool:
275 return bool(self.participant_column)
277 @property
278 def key_columns(self) -> tuple[str, ...]:
279 return (
280 ("participant_id", "trial_id")
281 if self.keyed_by_participant
282 else ("trial_id",)
283 )
285 @property
286 def names(self) -> tuple[str, ...]:
287 return tuple(field.name for field in self.fields)
289 def field(self, name: str) -> MetadataField | None:
290 for candidate in self.fields:
291 if candidate.name == name:
292 return candidate
293 return None
295 def key_series(self) -> pd.Series:
296 """The frame's own keys, as the string tuples the reports speak in."""
297 if self.frame.empty:
298 return pd.Series(dtype="object")
299 if self.keyed_by_participant:
300 return pd.Series(
301 list(
302 zip(
303 self.frame["participant_id"].astype(str),
304 self.frame["trial_id"].astype(str),
305 )
306 ),
307 index=self.frame.index,
308 )
309 return self.frame["trial_id"].astype(str)
311 @property
312 def joined_frame(self) -> pd.DataFrame:
313 """Only the rows describing trials that are actually loaded.
315 What the *controls* are built from, for `ParticipantMetadata`'s reason:
316 offering a value that belongs to a trial the report has just called "not
317 loaded — ignored" gives the user a filter that can only empty the pool.
318 """
319 if self.frame.empty:
320 return self.frame
321 return self.frame[self.key_series().isin(set(self.report.matched))]
323 def values_for(self, participant_id, trial_id) -> dict[str, object]:
324 """Every registered value for one reading (an unknown key gives ``{}``)."""
325 if self.frame.empty:
326 return {}
327 match = self.frame[self.frame["trial_id"] == str(trial_id)]
328 if self.keyed_by_participant:
329 match = match[match["participant_id"] == str(participant_id)]
330 if match.empty:
331 return {}
332 row = match.iloc[0]
333 return {name: row[name] for name in self.names if name in match.columns}
336@dataclass(frozen=True)
337class TextMetadata:
338 """A validated text table plus its field registry — the third grain.
340 ``frame`` always carries a string ``text_id`` column; conflicting ids have
341 been dropped from it (and named in :attr:`report`), the same rule
342 :class:`ParticipantMetadata`/:class:`TrialMetadata` follow. Flat grain —
343 one row per text, joined the way :class:`ParticipantMetadata` joins by
344 reader, never :class:`TrialMetadata`'s participant-pairing option: a text
345 is a stimulus, not something one reader owns.
346 """
348 frame: pd.DataFrame
349 fields: tuple[MetadataField, ...]
350 source_name: str
351 text_column: str
352 report: JoinReport = JoinReport()
354 @property
355 def names(self) -> tuple[str, ...]:
356 return tuple(field.name for field in self.fields)
358 def field(self, name: str) -> MetadataField | None:
359 for candidate in self.fields:
360 if candidate.name == name:
361 return candidate
362 return None
364 def values_for(self, text_id) -> dict[str, object]:
365 """Every registered value for one text (an unknown id gives ``{}``)."""
366 if self.frame.empty:
367 return {}
368 match = self.frame[self.frame["text_id"] == str(text_id)]
369 if match.empty:
370 return {}
371 row = match.iloc[0]
372 return {name: row[name] for name in self.names if name in match.columns}
374 @property
375 def joined_frame(self) -> pd.DataFrame:
376 """Only the rows describing texts that are actually loaded.
378 Same reasoning as :attr:`ParticipantMetadata.joined_frame` — the
379 controls must be built from what is on screen, not from every text
380 the table happens to mention.
381 """
382 if self.frame.empty:
383 return self.frame
384 return self.frame[self.frame["text_id"].isin(set(self.report.matched))]
386 def series(self, name: str) -> pd.Series:
387 """``text_id`` → value for one field, for projection/lookup."""
388 if self.frame.empty or name not in self.frame.columns:
389 return pd.Series(dtype="object")
390 return self.frame.set_index("text_id")[name]
393def _rows_with_ids(frame: pd.DataFrame, columns) -> pd.DataFrame:
394 """A copy of ``frame`` without the rows that have no value in an id column.
396 A blank row — the one Excel leaves at the end of a sheet — became a phantom
397 reader named "nan": under pandas 3 a missing id stays NaN through
398 ``stable_id``, and the ``!= ""`` test that used to drop it let NaN
399 through (BUG-60). A composite id with a missing part raised in the join
400 instead. Such a row describes no one, so it goes.
401 """
402 ids = frame[list(columns)]
403 missing = ids.isna() | ids.apply(lambda c: c.astype(str).str.strip() == "")
404 return frame.loc[~missing.any(axis=1)].copy()
407def _merge_duplicates(work: pd.DataFrame, key: pd.Series, value_columns) -> tuple:
408 """Fold the rows that repeat a key, unless they disagree.
410 Returns ``(work, key, duplicated, conflicting, combined_rows)``: ``work``
411 and ``key`` with one row per key, the keys that repeated, the keys whose
412 rows disagree (dropped whole, as every grain here has always done), and how
413 many rows were folded together.
415 Rows *disagree* when a field holds two different non-missing values. Rows
416 that do not are one record written twice, and each field takes the one
417 value they hold — two ``p1`` rows, one with a language and one with an age,
418 become one reader with both. Keeping the first row instead lost the second
419 row's values, and which ones depended on the file's row order.
420 """
421 repeated = key.duplicated(keep=False)
422 if not repeated.any():
423 return work, key, (), set(), 0
424 duplicated = tuple(sorted(set(key[repeated]), key=str))
425 columns = [column for column in value_columns if column in work.columns]
426 conflicting: set = set()
427 folded: list = []
428 combined = 0
429 # A positional index, so writing the folded values into a group's first row
430 # cannot also land on another row a user's frame gave the same label.
431 work, key = work.reset_index(drop=True), key.reset_index(drop=True)
432 repeated = repeated.reset_index(drop=True)
433 for group_key, rows in work[repeated].groupby(key[repeated], sort=False):
434 if any(rows[column].dropna().nunique() > 1 for column in columns):
435 conflicting.add(group_key)
436 continue
437 first = rows.index[0]
438 for column in columns:
439 present = rows[column].dropna()
440 if not present.empty:
441 work.at[first, column] = present.iloc[0]
442 folded.extend(rows.index[1:])
443 combined += len(rows)
444 keep = ~key.isin(conflicting) & ~work.index.isin(folded)
445 return work[keep], key[keep], duplicated, conflicting, combined
448def _range_mask(
449 frame: pd.DataFrame,
450 ranges: Mapping[str, tuple[float, float]],
451 keep_unknown: Mapping[str, bool] | None,
452) -> pd.Series:
453 """Rows inside every range in ``ranges`` (inclusive).
455 A row with no value for a ranged field is kept — a range narrows, it does
456 not exclude the unmeasured (UX-49) — unless ``keep_unknown`` maps that
457 field to ``False``: the researcher's explicit "only records with a
458 measured value".
459 """
460 mask = pd.Series(True, index=frame.index)
461 for name, (low, high) in ranges.items():
462 numeric = pd.to_numeric(frame[name], errors="coerce")
463 inside = numeric.between(low, high)
464 if (keep_unknown or {}).get(name, True) is not False:
465 inside |= numeric.isna()
466 mask &= inside
467 return mask
470def _keeps_unlisted(
471 ranges: Mapping[str, tuple[float, float]],
472 keep_unknown: Mapping[str, bool] | None,
473) -> bool:
474 """Whether a record the table has **no row** for survives ``ranges``.
476 It has no value for any field, so it is unknown for every range: kept
477 while every active range keeps its unknowns, left out once one does not.
478 """
479 return all((keep_unknown or {}).get(name, True) is not False for name in ranges)
482def numeric_extent(metadata, name: str) -> tuple[float, float] | None:
483 """``(min, max)`` of a numeric field over the loaded records, **equal ends
484 included** — or ``None`` when no loaded record has a value.
486 :func:`bounds_for` and its siblings answer "is there a range to slide
487 over?" and so return ``None`` for a constant field. This one answers "what
488 values are there?", which a constant field still has: the filter panel
489 shows it as a fixed value, and its *Keep unknown values* choice can still
490 leave out the records that have none. Works on any of the three tables.
491 """
492 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
493 return None
494 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna()
495 if numeric.empty:
496 return None
497 return float(numeric.min()), float(numeric.max())
500def unknown_count(metadata, name: str, keys: Iterable | None = None) -> int:
501 """How many loaded records have no value for ``name`` — the unknowns a
502 range keeps or, with *Keep unknown values* off, leaves out.
504 Counted in the unit the filter keeps or drops: readers for the participant
505 table, texts for the text table, and for the trial table **readings**
506 (``(participant_id, trial_id)`` pairs) when ``keys`` — the loaded pool's
507 pairs — is given, the table's own keys otherwise. A record with no row at
508 all is unknown too.
509 """
510 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
511 return 0
512 if isinstance(metadata, TrialMetadata) and keys is not None:
513 loaded = {tuple(str(part) for part in key) for key in keys}
514 values = pd.to_numeric(metadata.frame[name], errors="coerce")
515 known = set(metadata.key_series()[values.notna()])
516 if metadata.keyed_by_participant:
517 return sum(1 for key in loaded if key not in known)
518 return sum(1 for key in loaded if len(key) < 2 or key[1] not in known)
519 values = pd.to_numeric(metadata.joined_frame[name], errors="coerce")
520 return int(values.isna().sum()) + len(metadata.report.only_in_data)
523def active_trials() -> TrialMetadata | None:
524 """The trial table attached to this session, or ``None`` (DATA-29)."""
525 try:
526 import streamlit as st
528 return st.session_state.get(TRIAL_SESSION_KEY)
529 except Exception: # no script run context (API, CLI, plain import)
530 return None
533def trial_keys(combos: pd.DataFrame | None) -> set:
534 """The ``(participant_id, trial_id)`` pairs the loaded data actually has."""
535 if combos is None or combos.empty:
536 return set()
537 if not {"participant_id", "trial_id"} <= set(combos.columns):
538 return set()
539 pairs = combos[["participant_id", "trial_id"]].astype(str).drop_duplicates()
540 return set(map(tuple, pairs.to_numpy()))
543def infer_trial_id_column(frame: pd.DataFrame) -> str | None:
544 """First plausible trial-id column, or ``None`` — the UI's initial guess."""
545 if frame is None or frame.empty:
546 return None
547 lookup = {str(column).lower(): str(column) for column in frame.columns}
548 for candidate in TRIAL_ID_CANDIDATES:
549 hit = lookup.get(candidate.lower())
550 if hit is not None:
551 return hit
552 return None
555def _trial_column_label(trial_column) -> str:
556 """Display form of a (possibly composite) trial-column mapping — joined
557 with " + ", matching how the wizard spells a composite trial id back out."""
558 return " + ".join(trial_mapping_columns(trial_column))
561def build_trial_metadata(
562 frame: pd.DataFrame,
563 trial_column: str | list[str],
564 participant_column: str | None = None,
565 *,
566 source_name: str = "trial metadata",
567 keys: Iterable | None = None,
568) -> TrialMetadata:
569 """Validate a raw trial table into a registry + clean frame (DATA-29).
571 ``trial_column`` is a single column name, or **several** to build a unique
572 trial id on the fly (joined with ``_``, like the Trial ID mapping the
573 uploaded data itself uses — see :func:`data.trial_id_series`) — for a
574 table whose own trial id needs the same composite key the data does.
576 The participant half of the key is **optional and explicit**: pass
577 ``participant_column`` to key by reader *and* trial. ``keys`` is the set of
578 ``(participant_id, trial_id)`` pairs present in the loaded data, which fills
579 in the two "unmatched" halves of the report — and, when the table is keyed by
580 trial alone, is collapsed to trial ids first so a table that legitimately
581 describes one reading per text is not reported as missing every reader.
583 Mirrors :func:`build_participant_metadata` deliberately, including the rule
584 that duplicate rows are only a problem when they *disagree*.
585 """
586 trial_cols = trial_mapping_columns(trial_column)
587 label = _trial_column_label(trial_column)
588 empty = TrialMetadata(
589 pd.DataFrame(columns=["trial_id"]),
590 (),
591 source_name,
592 label,
593 str(participant_column) if participant_column else None,
594 )
595 if (
596 frame is None
597 or frame.empty
598 or not trial_cols
599 or any(c not in frame.columns for c in trial_cols)
600 ):
601 return empty
602 if participant_column and participant_column not in frame.columns:
603 participant_column = None
605 work = _rows_with_ids(
606 frame, [*trial_cols, *([participant_column] if participant_column else [])]
607 )
608 # `trial_id_series` — not a plain `.astype(str)` — so this table's own
609 # trial id is spelled the same way `data.normalize_*` spells the app's: a
610 # blank cell anywhere else in *this* file's trial-id column is enough to
611 # read it as floats ("101.0") against the data's "101", and the join below
612 # would silently match nothing (DATA-29's "no reading matched" is exactly
613 # this) — and a composite id is built the identical way (`compose_id`,
614 # each part through `stable_id` first).
615 work["trial_id"] = trial_id_series(work, trial_column)
616 if participant_column:
617 work["participant_id"] = stable_id(work[participant_column])
618 reserved = {
619 *trial_cols,
620 str(participant_column) if participant_column else "",
621 "trial_id",
622 "participant_id",
623 *_BOOKKEEPING_COLUMNS,
624 }
625 value_columns = [
626 str(column) for column in frame.columns if str(column) not in reserved
627 ]
629 key_frame = (
630 pd.Series(list(zip(work["participant_id"], work["trial_id"])), index=work.index)
631 if participant_column
632 else work["trial_id"]
633 )
634 work, key_frame, duplicated, conflicting_set, combined = _merge_duplicates(
635 work, key_frame, value_columns
636 )
638 clean = pd.DataFrame({"trial_id": work["trial_id"].to_numpy()})
639 if participant_column:
640 clean.insert(0, "participant_id", work["participant_id"].to_numpy())
641 fields: list[MetadataField] = []
642 for column in value_columns:
643 dtype = _classify(work[column])
644 values = _coerce(work[column], dtype)
645 clean[column] = values.to_numpy()
646 fields.append(
647 MetadataField(
648 name=column,
649 label=field_label(column),
650 grain=GRAIN_TRIAL,
651 dtype=dtype,
652 source=source_name,
653 n_unique=int(values.dropna().nunique()),
654 n_missing=int(values.isna().sum()),
655 )
656 )
658 metadata = TrialMetadata(
659 clean,
660 tuple(fields),
661 source_name,
662 label,
663 str(participant_column) if participant_column else None,
664 JoinReport(
665 matched=tuple(sorted(set(key_frame), key=str)),
666 duplicated=tuple(sorted(duplicated, key=str)),
667 conflicting=tuple(sorted(conflicting_set, key=str)),
668 combined_rows=combined,
669 ),
670 )
671 if keys is None:
672 return metadata
673 return rejoin_trials(metadata, keys)
676def rejoin_trials(metadata: TrialMetadata, keys: Iterable) -> TrialMetadata:
677 """Recompute the join report against the trials actually loaded (DATA-29).
679 ``keys`` are ``(participant_id, trial_id)`` pairs; a table keyed by trial
680 alone is compared on the trial half, so "this file describes texts, not
681 readings" is a supported answer rather than a report full of misses.
682 """
683 data_keys = {tuple(str(part) for part in key) for key in keys}
684 if not metadata.keyed_by_participant:
685 data_keys = {key[1] for key in data_keys if len(key) > 1}
686 if not metadata.frame.empty:
687 data_trials = {
688 key[1] if isinstance(key, tuple) else key
689 for key in data_keys
690 if not isinstance(key, tuple) or len(key) > 1
691 }
692 respelled = _respell_ids(metadata.frame["trial_id"], data_trials)
693 if not respelled.equals(metadata.frame["trial_id"]):
694 metadata = replace(
695 metadata, frame=metadata.frame.assign(trial_id=respelled)
696 )
697 table_keys = set(metadata.key_series()) | set(metadata.report.conflicting)
698 usable = set(metadata.key_series())
699 return TrialMetadata(
700 metadata.frame,
701 metadata.fields,
702 metadata.source_name,
703 metadata.trial_column,
704 metadata.participant_column,
705 JoinReport(
706 matched=tuple(sorted(usable & data_keys, key=str)),
707 only_in_table=tuple(sorted(table_keys - data_keys, key=str)),
708 only_in_data=tuple(sorted(data_keys - table_keys, key=str)),
709 duplicated=metadata.report.duplicated,
710 conflicting=metadata.report.conflicting,
711 combined_rows=metadata.report.combined_rows,
712 ),
713 )
716def trials_matching(
717 metadata: TrialMetadata | None,
718 selections: dict[str, Sequence] | None = None,
719 ranges: dict[str, tuple[float, float]] | None = None,
720 *,
721 keys: Iterable | None = None,
722 keep_unknown: Mapping[str, bool] | None = None,
723) -> set | None:
724 """``(participant_id, trial_id)`` keys satisfying every trial constraint.
726 ``None`` means "no constraint" — the same contract as
727 :func:`participants_matching`, and for the same reason: an empty selection
728 must not narrow the pool to the trials the table happens to list.
730 ``keys`` are the loaded trials, needed for two things a trial-grain table
731 cannot do without: expanding a trial-id-keyed table back to the readings
732 that share that trial id, and keeping the trials the table never mentions
733 when the only constraint is a numeric range (``data.filter_trials``' rule
734 that a range narrows rather than excludes the unmeasured — UX-49).
736 ``keep_unknown`` maps a ranged field to ``False`` to leave its unknowns
737 out instead: a reading whose value is missing, and one with no row at all
738 (see :func:`_range_mask`).
739 """
740 if metadata is None or metadata.frame.empty:
741 return None
742 active_selections = {
743 name: list(values)
744 for name, values in (selections or {}).items()
745 if values and name in metadata.frame.columns
746 }
747 active_ranges = {
748 name: bounds
749 for name, bounds in (ranges or {}).items()
750 if bounds and name in metadata.frame.columns
751 }
752 if not active_selections and not active_ranges:
753 return None
755 frame = metadata.frame
756 mask = pd.Series(True, index=frame.index)
757 for name, values in active_selections.items():
758 allowed = {str(value) for value in values}
759 mask &= frame[name].astype(str).isin(allowed)
760 mask &= _range_mask(frame, active_ranges, keep_unknown)
761 matching = set(metadata.key_series()[mask])
763 loaded = {tuple(str(part) for part in key) for key in (keys or ())}
764 if metadata.keyed_by_participant:
765 result = {key for key in loaded if key in matching} if loaded else set(matching)
766 else:
767 # Trial-id grain describes every reading of that trial.
768 result = {key for key in loaded if key[1] in matching}
769 if (
770 not active_selections
771 and loaded
772 and _keeps_unlisted(active_ranges, keep_unknown)
773 ):
774 # Range-only narrowing keeps the unmeasured, including a reading with no
775 # row at all — the participant table's rule, one grain down. "Has a
776 # row" is read from the whole table, never from the rows that passed
777 # the range: a reading described *outside* the range is described, and
778 # treating it as unlisted added every one of them back.
779 listed = set(metadata.key_series())
780 described = (
781 listed
782 if metadata.keyed_by_participant
783 else {key for key in loaded if key[1] in listed}
784 )
785 result |= {key for key in loaded if key not in described}
786 return result
789def project_trials(
790 metadata: TrialMetadata | None,
791 frame: pd.DataFrame,
792 columns: Iterable[str] | None = None,
793) -> pd.DataFrame:
794 """Left-join chosen trial-metadata columns onto a **small** trial-keyed frame.
796 For ``combos`` (one row per trial) — never for words or fixations, which is
797 the rule the participant table follows for the same reason. Columns already
798 on ``frame`` win, so a recorded column is never shadowed.
799 """
800 if (
801 metadata is None
802 or metadata.frame.empty
803 or frame is None
804 or frame.empty
805 or "trial_id" not in frame.columns
806 ):
807 return frame
808 if metadata.keyed_by_participant and "participant_id" not in frame.columns:
809 return frame
810 wanted = [
811 name
812 for name in (list(columns) if columns is not None else list(metadata.names))
813 if name in metadata.frame.columns and name not in frame.columns
814 ]
815 if not wanted:
816 return frame
817 out = frame.copy()
818 if metadata.keyed_by_participant:
819 keys = pd.Series(
820 list(zip(out["participant_id"].astype(str), out["trial_id"].astype(str))),
821 index=out.index,
822 )
823 else:
824 keys = out["trial_id"].astype(str)
825 lookup = metadata.frame.set_index(metadata.key_series())
826 for name in wanted:
827 out[name] = keys.map(lookup[name])
828 return out
831def trial_options_for(metadata: TrialMetadata | None, name: str) -> list[str]:
832 """Sorted distinct values of a categorical trial field, for a multiselect."""
833 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
834 return []
835 return sorted({str(value) for value in metadata.joined_frame[name].dropna()})
838def trial_bounds_for(
839 metadata: TrialMetadata | None, name: str
840) -> tuple[float, float] | None:
841 """``(min, max)`` of a numeric trial field over the loaded trials, or
842 ``None`` when it has no range.
844 A field with one value over the loaded trials — a one-row table, or a pilot
845 whose trials all share it — has no range, the rule :func:`bounds_for` and
846 :func:`text_bounds_for` already follow. Returning ``(20.0, 20.0)`` handed
847 the filter panel a slider Streamlit refuses to draw, which stopped the
848 whole Scanpath view.
849 """
850 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
851 return None
852 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna()
853 if numeric.empty:
854 return None
855 low, high = float(numeric.min()), float(numeric.max())
856 if low == high:
857 return None
858 return low, high
861def trial_to_payload(metadata: TrialMetadata | None) -> dict | None:
862 """Serialize the trial table for save & restore (DATA-29)."""
863 if metadata is None or metadata.frame.empty:
864 return None
865 return {
866 "source_name": metadata.source_name,
867 "trial_column": metadata.trial_column,
868 "participant_column": metadata.participant_column,
869 "rows": metadata.frame.to_dict("records"),
870 }
873def trial_from_payload(payload: dict | None) -> TrialMetadata | None:
874 """Rebuild a trial table from :func:`trial_to_payload`'s output."""
875 if not isinstance(payload, dict) or not payload.get("rows"):
876 return None
877 frame = pd.DataFrame(payload["rows"])
878 trial_column = str(payload.get("trial_column") or "trial_id")
879 participant_column = payload.get("participant_column")
880 if "trial_id" in frame.columns:
881 # The payload holds the *clean* frame, whose key columns are already
882 # canonical — rebuild against those rather than the original names.
883 return build_trial_metadata(
884 frame,
885 "trial_id",
886 "participant_id" if participant_column else None,
887 source_name=str(payload.get("source_name") or "trial metadata"),
888 )
889 return build_trial_metadata(
890 frame,
891 trial_column,
892 participant_column,
893 source_name=str(payload.get("source_name") or "trial metadata"),
894 )
897def active_texts() -> TextMetadata | None:
898 """The text table attached to this session, or ``None``."""
899 try:
900 import streamlit as st
902 return st.session_state.get(TEXT_SESSION_KEY)
903 except Exception: # no script run context (API, CLI, plain import)
904 return None
907def text_keys(combos: pd.DataFrame | None) -> set:
908 """The distinct ``text_id`` values the loaded data actually has."""
909 if combos is None or combos.empty or "text_id" not in combos.columns:
910 return set()
911 return {str(value) for value in combos["text_id"].dropna().unique()}
914def infer_text_id_column(frame: pd.DataFrame) -> str | None:
915 """First plausible text-id column, or ``None`` — the UI's initial guess."""
916 if frame is None or frame.empty:
917 return None
918 lookup = {str(column).lower(): str(column) for column in frame.columns}
919 for candidate in TEXT_ID_CANDIDATES:
920 hit = lookup.get(candidate.lower())
921 if hit is not None:
922 return hit
923 return None
926def _text_column_label(text_column) -> str:
927 """Display form of a (possibly composite) text-column mapping — joined
928 with " + ", matching how the wizard spells a composite trial id back out."""
929 return " + ".join(trial_mapping_columns(text_column))
932def build_text_metadata(
933 frame: pd.DataFrame,
934 text_column: str | list[str],
935 *,
936 source_name: str = "text metadata",
937 keys: Iterable | None = None,
938) -> TextMetadata:
939 """Validate a raw text table into a registry + clean frame — third grain.
941 ``text_column`` is a single column name, or **several** to build a unique
942 text id on the fly (joined with ``_``, like the Trial ID mapping the
943 uploaded data itself uses — see :func:`data.trial_id_series`), the same
944 composite-key trick :func:`build_trial_metadata` uses. The join/report
945 logic underneath is flat, though — one dimension (``text_id``), never
946 :func:`build_trial_metadata`'s participant-pairing option, since a text is
947 a stimulus and nothing reads it as belonging to one reader.
949 ``keys`` is the set of text ids actually present in the loaded data;
950 passing it fills in the two "unmatched" halves of the report. Rows whose
951 id repeats are only a problem when they *disagree* — the rule every grain
952 here follows.
953 """
954 text_cols = trial_mapping_columns(text_column)
955 label = _text_column_label(text_column)
956 empty = TextMetadata(pd.DataFrame(columns=["text_id"]), (), source_name, label)
957 if (
958 frame is None
959 or frame.empty
960 or not text_cols
961 or any(c not in frame.columns for c in text_cols)
962 ):
963 return empty
965 work = _rows_with_ids(frame, text_cols)
966 # See the matching comment in `build_trial_metadata` — the same "one
967 # blank cell spells the id two ways" hazard applies to a text id.
968 work["text_id"] = trial_id_series(work, text_column)
969 if keys is not None:
970 keys = list(keys)
971 work["text_id"] = _respell_ids(work["text_id"], {str(tid) for tid in keys})
972 reserved = {*text_cols, "text_id", *_BOOKKEEPING_COLUMNS}
973 value_columns = [
974 str(column) for column in frame.columns if str(column) not in reserved
975 ]
977 work, _, duplicated, conflicting_set, combined = _merge_duplicates(
978 work, work["text_id"], value_columns
979 )
981 fields: list[MetadataField] = []
982 clean = pd.DataFrame({"text_id": work["text_id"].to_numpy()})
983 for column in value_columns:
984 dtype = _classify(work[column])
985 values = _coerce(work[column], dtype)
986 clean[column] = values.to_numpy()
987 fields.append(
988 MetadataField(
989 name=column,
990 label=field_label(column),
991 grain=GRAIN_TEXT,
992 dtype=dtype,
993 source=source_name,
994 n_unique=int(values.dropna().nunique()),
995 n_missing=int(values.isna().sum()),
996 )
997 )
999 table_ids = set(clean["text_id"]) | conflicting_set
1000 if keys is None:
1001 report = JoinReport(
1002 matched=tuple(sorted(clean["text_id"])),
1003 duplicated=duplicated,
1004 conflicting=tuple(sorted(conflicting_set)),
1005 combined_rows=combined,
1006 )
1007 else:
1008 data_ids = {str(tid) for tid in keys}
1009 report = JoinReport(
1010 matched=tuple(sorted(set(clean["text_id"]) & data_ids)),
1011 only_in_table=tuple(sorted(table_ids - data_ids)),
1012 only_in_data=tuple(sorted(data_ids - table_ids)),
1013 duplicated=duplicated,
1014 conflicting=tuple(sorted(conflicting_set)),
1015 combined_rows=combined,
1016 )
1017 return TextMetadata(clean, tuple(fields), source_name, label, report)
1020def texts_matching(
1021 metadata: TextMetadata | None,
1022 selections: dict[str, Sequence] | None = None,
1023 ranges: dict[str, tuple[float, float]] | None = None,
1024 *,
1025 keep_unknown: Mapping[str, bool] | None = None,
1026) -> set | None:
1027 """Text ids satisfying every metadata constraint, or ``None`` for "any".
1029 Flat-grain sibling of :func:`participants_matching` — an empty constraint
1030 must not narrow the pool to the texts *listed in the table*, and a numeric
1031 range keeps a text with no value (``data.filter_trials``' rule that a
1032 range narrows rather than excludes the unmeasured) unless ``keep_unknown``
1033 maps that field to ``False``.
1034 """
1035 if metadata is None or metadata.frame.empty:
1036 return None
1037 active = {
1038 name: list(values)
1039 for name, values in (selections or {}).items()
1040 if values and name in metadata.frame.columns
1041 }
1042 active_ranges = {
1043 name: bounds
1044 for name, bounds in (ranges or {}).items()
1045 if bounds and name in metadata.frame.columns
1046 }
1047 if not active and not active_ranges:
1048 return None
1050 frame = metadata.frame
1051 mask = pd.Series(True, index=frame.index)
1052 for name, values in active.items():
1053 allowed = {str(value) for value in values}
1054 mask &= frame[name].astype(str).isin(allowed)
1055 mask &= _range_mask(frame, active_ranges, keep_unknown)
1056 matching = set(frame.loc[mask, "text_id"])
1057 if not active and _keeps_unlisted(active_ranges, keep_unknown):
1058 # Range-only narrowing keeps the unmeasured, including a text with no
1059 # row at all (`participants_matching`'s rule, one grain over).
1060 matching |= set(metadata.report.only_in_data)
1061 return matching
1064def project_texts(
1065 metadata: TextMetadata | None,
1066 frame: pd.DataFrame,
1067 columns: Iterable[str] | None = None,
1068) -> pd.DataFrame:
1069 """Left-join chosen text-metadata columns onto a **small** text-keyed frame.
1071 For ``combos`` — never for words or fixations, the rule every grain here
1072 follows. Columns already present on ``frame`` win, so a recorded column is
1073 never shadowed by a metadata field of the same name.
1074 """
1075 if (
1076 metadata is None
1077 or metadata.frame.empty
1078 or frame is None
1079 or frame.empty
1080 or "text_id" not in frame.columns
1081 ):
1082 return frame
1083 wanted = [
1084 name
1085 for name in (list(columns) if columns is not None else list(metadata.names))
1086 if name in metadata.frame.columns and name not in frame.columns
1087 ]
1088 if not wanted:
1089 return frame
1090 out = frame.copy()
1091 keys = out["text_id"].astype(str)
1092 for name in wanted:
1093 out[name] = keys.map(metadata.series(name))
1094 return out
1097def text_options_for(metadata: TextMetadata | None, name: str) -> list[str]:
1098 """Sorted distinct values of a categorical text field, for a multiselect."""
1099 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
1100 return []
1101 return sorted({str(value) for value in metadata.joined_frame[name].dropna()})
1104def text_bounds_for(
1105 metadata: TextMetadata | None, name: str
1106) -> tuple[float, float] | None:
1107 """``(min, max)`` of a numeric text field, or ``None`` when it has no range."""
1108 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
1109 return None
1110 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna()
1111 if numeric.empty:
1112 return None
1113 low, high = float(numeric.min()), float(numeric.max())
1114 if low == high:
1115 return None
1116 return low, high
1119def text_to_payload(metadata: TextMetadata | None) -> dict | None:
1120 """Serialize the text table for save & restore."""
1121 if metadata is None or metadata.frame.empty:
1122 return None
1123 return {
1124 "source_name": metadata.source_name,
1125 "text_column": metadata.text_column,
1126 "rows": metadata.frame.to_dict("records"),
1127 }
1130def text_from_payload(payload: dict | None) -> TextMetadata | None:
1131 """Rebuild a text table from :func:`text_to_payload`'s output."""
1132 if not isinstance(payload, dict) or not payload.get("rows"):
1133 return None
1134 frame = pd.DataFrame(payload["rows"])
1135 text_column = str(payload.get("text_column") or "text_id")
1136 if "text_id" in frame.columns:
1137 # The payload holds the *clean* frame, whose key column is already
1138 # canonical — rebuild against that rather than the original name.
1139 return build_text_metadata(
1140 frame,
1141 "text_id",
1142 source_name=str(payload.get("source_name") or "text metadata"),
1143 )
1144 return build_text_metadata(
1145 frame,
1146 text_column,
1147 source_name=str(payload.get("source_name") or "text metadata"),
1148 )
1151def active() -> ParticipantMetadata | None:
1152 """The participant table attached to this session, or ``None``.
1154 Lives here rather than in the UI layer so the pure consumers
1155 (:func:`project`, the filter resolution in ``controls``) can reach it
1156 without importing Streamlit page code. Returns ``None`` outside a script
1157 run, which is what the headless API and CLI see.
1158 """
1159 try:
1160 import streamlit as st
1162 return st.session_state.get(SESSION_KEY)
1163 except Exception: # no script run context (API, CLI, plain import)
1164 return None
1167def participant_ids(*frames: pd.DataFrame | None) -> list[str]:
1168 """Every distinct reader id across the given frames, as sorted strings."""
1169 found: set = set()
1170 for frame in frames:
1171 if frame is None or frame.empty or "participant_id" not in frame.columns:
1172 continue
1173 found |= {str(value) for value in frame["participant_id"].dropna().unique()}
1174 return sorted(found)
1177def infer_participant_id_column(frame: pd.DataFrame) -> str | None:
1178 """Best guess at the reader-id column, or ``None`` when nothing fits."""
1179 if frame is None or frame.empty:
1180 return None
1181 lowered = {str(column).lower(): str(column) for column in frame.columns}
1182 for candidate in PARTICIPANT_ID_CANDIDATES:
1183 if candidate in frame.columns:
1184 return candidate
1185 hit = lowered.get(candidate.lower())
1186 if hit is not None:
1187 return hit
1188 return None
1191def _classify(series: pd.Series) -> str:
1192 """Dtype bucket driving which control a field gets (range vs membership)."""
1193 cleaned = series.dropna()
1194 if cleaned.empty:
1195 return _DTYPE_CATEGORICAL
1196 if pd.api.types.is_bool_dtype(cleaned):
1197 return _DTYPE_BOOLEAN
1198 if pd.api.types.is_numeric_dtype(cleaned):
1199 return _DTYPE_NUMERIC
1200 # A column of numeric strings ("23", "4.5") is numeric in every way the user
1201 # cares about; anything else stays categorical rather than being coerced.
1202 numeric = pd.to_numeric(cleaned, errors="coerce")
1203 if numeric.notna().all():
1204 return _DTYPE_NUMERIC
1205 return _DTYPE_CATEGORICAL
1208def _coerce(series: pd.Series, dtype: str) -> pd.Series:
1209 if dtype == _DTYPE_NUMERIC:
1210 # An infinite value is no value (round 10): the record counts as
1211 # unknown, which a range keeps unless *Keep unknown values* is off —
1212 # and a slider never gets an infinite end. All three grains read here.
1213 numbers = pd.to_numeric(series, errors="coerce")
1214 infinite = np.isinf(numbers.astype(float))
1215 return numbers.mask(infinite) if infinite.any() else numbers
1216 if dtype == _DTYPE_BOOLEAN:
1217 return series
1218 return series.astype("object").where(series.notna(), np.nan)
1221def field_label(name: str) -> str:
1222 """The label for a metadata field: its column name, as the table spelled it.
1224 DATA-66: a field the user attached is shown under the name it has in their
1225 file (``native_language``, not "Native language"). Public because it is the
1226 *only* labeller for a metadata field: the picker in ``tabs._pretty_col`` has
1227 to name a field the same way whether or not it can reach the attached table
1228 at that moment.
1229 """
1230 return str(name)
1233def build_participant_metadata(
1234 frame: pd.DataFrame,
1235 id_column: str,
1236 *,
1237 source_name: str = "participant metadata",
1238 participants: Iterable | None = None,
1239) -> ParticipantMetadata:
1240 """Validate a raw participant table into a registry + clean frame.
1242 ``participants`` is the set of reader ids actually present in the loaded
1243 data; passing it fills in the two "unmatched" halves of the report. Rows
1244 whose id repeats are only a problem when they *disagree* — duplicated rows
1245 that do not are combined field by field (:func:`_merge_duplicates`, counted
1246 in ``report.combined_rows``), duplicated rows that say something different
1247 are dropped and reported.
1248 """
1249 if frame is None or frame.empty or id_column not in frame.columns:
1250 return ParticipantMetadata(
1251 pd.DataFrame(columns=["participant_id"]), (), source_name, str(id_column)
1252 )
1254 work = _rows_with_ids(frame, [id_column])
1255 # See the matching comment in `build_trial_metadata` — the same "one blank
1256 # cell spells the id two ways" hazard applies to a reader id.
1257 work["participant_id"] = stable_id(work[id_column])
1258 if participants is not None:
1259 work["participant_id"] = _match_padding(work["participant_id"], participants)
1260 value_columns = [
1261 str(column)
1262 for column in frame.columns
1263 if str(column) not in {str(id_column), "participant_id", *_BOOKKEEPING_COLUMNS}
1264 ]
1266 work, _, duplicated, conflicting_set, combined = _merge_duplicates(
1267 work, work["participant_id"], value_columns
1268 )
1270 fields: list[MetadataField] = []
1271 clean = pd.DataFrame({"participant_id": work["participant_id"].to_numpy()})
1272 for column in value_columns:
1273 dtype = _classify(work[column])
1274 values = _coerce(work[column], dtype)
1275 clean[column] = values.to_numpy()
1276 fields.append(
1277 MetadataField(
1278 name=column,
1279 label=field_label(column),
1280 grain=GRAIN_PARTICIPANT,
1281 dtype=dtype,
1282 source=source_name,
1283 n_unique=int(values.dropna().nunique()),
1284 n_missing=int(values.isna().sum()),
1285 )
1286 )
1288 table_ids = set(clean["participant_id"]) | conflicting_set
1289 if participants is None:
1290 report = JoinReport(
1291 matched=tuple(sorted(clean["participant_id"])),
1292 duplicated=duplicated,
1293 conflicting=tuple(sorted(conflicting_set)),
1294 combined_rows=combined,
1295 )
1296 else:
1297 data_ids = {str(pid) for pid in participants}
1298 report = JoinReport(
1299 matched=tuple(sorted(set(clean["participant_id"]) & data_ids)),
1300 only_in_table=tuple(sorted(table_ids - data_ids)),
1301 only_in_data=tuple(sorted(data_ids - table_ids)),
1302 duplicated=duplicated,
1303 conflicting=tuple(sorted(conflicting_set)),
1304 combined_rows=combined,
1305 )
1306 return ParticipantMetadata(
1307 clean, tuple(fields), source_name, str(id_column), report
1308 )
1311def _match_padding(ids: pd.Series, participants: Iterable) -> pd.Series:
1312 """``ids`` spelled the data's way when only zero-padding differs (BUG-59).
1314 A metadata CSV reads a reader ``007`` as the number 7 while the data kept
1315 "007", and the table then joined to no one; ``data.zero_padding_map``
1316 decides, and refuses whenever the match is not unambiguous.
1317 """
1318 return _respell_ids(ids, {str(pid) for pid in participants})
1321def _respell_ids(ids: pd.Series, reference: set) -> pd.Series:
1322 """``ids`` spelled the way ``reference`` spells them, when the two differ
1323 only by zero-padding (BUG-59) or by the escaping composite ids gained
1324 (``data.composite_respelling_map``: a dataset restored from the recovery
1325 cache keeps the ids it was stored with, a table attached today composes
1326 them anew). Both maps refuse anything ambiguous."""
1327 unique = ids.unique()
1328 mapping = zero_padding_map(unique, reference) or composite_respelling_map(
1329 unique, reference
1330 )
1331 return ids.replace(mapping) if mapping else ids
1334def rejoin(
1335 metadata: ParticipantMetadata, participants: Iterable
1336) -> ParticipantMetadata:
1337 """Recompute the join report against a (possibly new) participant list."""
1338 data_ids = {str(pid) for pid in participants}
1339 if not metadata.frame.empty:
1340 renamed = _match_padding(metadata.frame["participant_id"], data_ids)
1341 if not renamed.equals(metadata.frame["participant_id"]):
1342 metadata = replace(
1343 metadata, frame=metadata.frame.assign(participant_id=renamed)
1344 )
1345 usable_ids = (
1346 set(metadata.frame["participant_id"]) if not metadata.frame.empty else set()
1347 )
1348 # Conflicting ids are *in the table* — so they are not "only in the data" —
1349 # but they carry no values, so they are not joined either. Counting them as
1350 # matched (as an earlier version did) made "Joined to N readers" grow by the
1351 # conflict count on the first rerun after the file was attached, disagreeing
1352 # with what `build_participant_metadata` had just reported.
1353 table_ids = usable_ids | set(metadata.report.conflicting)
1354 return ParticipantMetadata(
1355 metadata.frame,
1356 metadata.fields,
1357 metadata.source_name,
1358 metadata.id_column,
1359 JoinReport(
1360 matched=tuple(sorted(usable_ids & data_ids)),
1361 only_in_table=tuple(sorted(table_ids - data_ids)),
1362 only_in_data=tuple(sorted(data_ids - table_ids)),
1363 duplicated=metadata.report.duplicated,
1364 conflicting=metadata.report.conflicting,
1365 combined_rows=metadata.report.combined_rows,
1366 ),
1367 )
1370def participants_matching(
1371 metadata: ParticipantMetadata | None,
1372 selections: dict[str, Sequence] | None = None,
1373 ranges: dict[str, tuple[float, float]] | None = None,
1374 *,
1375 keep_unknown: Mapping[str, bool] | None = None,
1376) -> set | None:
1377 """Reader ids satisfying every metadata constraint, or ``None`` for "any".
1379 Returning ``None`` rather than "all ids" is deliberate: an empty constraint
1380 must not narrow the pool to the readers *listed in the table*, which would
1381 quietly drop everyone the table forgot.
1383 Membership follows the categorical filters; a numeric range keeps readers
1384 with **no value**, matching ``data.filter_trials``' rule that a range is a
1385 narrowing control and not an exclusion of the unmeasured. ``keep_unknown``
1386 maps a ranged field to ``False`` to leave those readers out instead — the
1387 explicit *Keep unknown values* choice beside the slider.
1388 """
1389 if metadata is None or metadata.frame.empty:
1390 return None
1391 active = {
1392 name: list(values)
1393 for name, values in (selections or {}).items()
1394 if values and name in metadata.frame.columns
1395 }
1396 active_ranges = {
1397 name: bounds
1398 for name, bounds in (ranges or {}).items()
1399 if bounds and name in metadata.frame.columns
1400 }
1401 if not active and not active_ranges:
1402 return None
1404 frame = metadata.frame
1405 mask = pd.Series(True, index=frame.index)
1406 for name, values in active.items():
1407 allowed = {str(value) for value in values}
1408 mask &= frame[name].astype(str).isin(allowed)
1409 mask &= _range_mask(frame, active_ranges, keep_unknown)
1410 matching = set(frame.loc[mask, "participant_id"])
1411 if not active and _keeps_unlisted(active_ranges, keep_unknown):
1412 # Range-only narrowing keeps the unmeasured (`data.filter_trials`' rule,
1413 # UX-49) — and a reader with **no row at all** is the most unmeasured
1414 # there is, so they are kept on the same terms as a reader whose value
1415 # is NaN. A *categorical* selection still excludes them, matching every
1416 # other membership filter in the app: "only Hebrew speakers" cannot
1417 # include a reader whose language is unknown.
1418 matching |= set(metadata.report.only_in_data)
1419 return matching
1422def project(
1423 metadata: ParticipantMetadata | None,
1424 frame: pd.DataFrame,
1425 columns: Iterable[str] | None = None,
1426) -> pd.DataFrame:
1427 """Left-join chosen metadata columns onto a **small** participant-keyed frame.
1429 For ``combos`` (one row per trial) and group-by results — never for words or
1430 fixations. Columns already present on ``frame`` win, so a real recorded
1431 column is never shadowed by a metadata field of the same name.
1432 """
1433 if (
1434 metadata is None
1435 or metadata.frame.empty
1436 or frame is None
1437 or frame.empty
1438 or "participant_id" not in frame.columns
1439 ):
1440 return frame
1441 wanted = [
1442 name
1443 for name in (list(columns) if columns is not None else list(metadata.names))
1444 if name in metadata.frame.columns and name not in frame.columns
1445 ]
1446 if not wanted:
1447 return frame
1448 out = frame.copy()
1449 keys = out["participant_id"].astype(str)
1450 for name in wanted:
1451 out[name] = keys.map(metadata.series(name))
1452 return out
1455def options_for(metadata: ParticipantMetadata | None, name: str) -> list[str]:
1456 """Sorted distinct values of a categorical field, for a multiselect.
1458 Built from :attr:`ParticipantMetadata.joined_frame` — the loaded readers
1459 only — so the control cannot offer a value that matches nobody.
1460 """
1461 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
1462 return []
1463 values = metadata.joined_frame[name].dropna()
1464 return sorted({str(value) for value in values})
1467def bounds_for(
1468 metadata: ParticipantMetadata | None, name: str
1469) -> tuple[float, float] | None:
1470 """``(min, max)`` of a numeric field, or ``None`` when it has no range."""
1471 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns:
1472 return None
1473 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna()
1474 if numeric.empty:
1475 return None
1476 low, high = float(numeric.min()), float(numeric.max())
1477 if low == high:
1478 return None
1479 return low, high
1482# -----------------------------------------------------------------------------
1483# Serialization — the ENG-26 on-device recovery cache (DATA-38, see
1484# `session_payloads` below) and the payload `api`/`cli` hand in. (The 💾 Session
1485# backup carried these too until UX-179 cut the settings file to the figure.) Records rather than a pickled frame, so it round-trips
1486# through JSON like every other saved setting.
1487# -----------------------------------------------------------------------------
1490def to_payload(metadata: ParticipantMetadata | None) -> dict | None:
1491 if metadata is None or metadata.frame.empty:
1492 return None
1493 return {
1494 "grain": GRAIN_PARTICIPANT,
1495 "id_column": metadata.id_column,
1496 "source_name": metadata.source_name,
1497 "records": metadata.frame.to_dict(orient="records"),
1498 }
1501def from_payload(payload: dict | None) -> ParticipantMetadata | None:
1502 if not payload or not payload.get("records"):
1503 return None
1504 frame = pd.DataFrame(payload["records"])
1505 if "participant_id" not in frame.columns:
1506 return None
1507 return build_participant_metadata(
1508 frame,
1509 "participant_id",
1510 source_name=str(payload.get("source_name") or "participant metadata"),
1511 )
1514# -----------------------------------------------------------------------------
1515# DATA-38 — attached tables in the ENG-26 on-device recovery cache, and DATA-47 —
1516# the tables belong to a dataset.
1517#
1518# The session keys above hold the tables of the *selected* dataset only — the
1519# one every consumer (filters, chips, sort, inspection, export) reads through
1520# `active()` / `active_trials()` / `active_texts()`. Every other dataset's
1521# tables wait in a per-dataset store, and `activate_dataset` swaps them in and
1522# out when the selection changes. They used to be one slot per grain for the
1523# whole session, so a new dataset opened with the last one's tables, attaching a
1524# table to dataset B replaced dataset A's, and detaching it anywhere removed it
1525# everywhere. The cache writes the store, keyed by dataset.
1526# -----------------------------------------------------------------------------
1528#: What a grain's ``*_FILE_SESSION_KEY`` holds when its table came back from the
1529#: recovery cache or a saved config, rather than from a file in the uploader.
1530#: The metadata sections read an empty uploader as "the user just removed the
1531#: file" and detach on sight (UX-115) — and a restored table has no file in the
1532#: uploader, so without this marker the first visit to the 🗂️ Data page would
1533#: detach exactly what the restore brought back.
1534RESTORED_FILE_SIGNATURE = "restored"
1536#: ``(grain, table key, raw key, file key, to_payload, from_payload)`` per grain.
1537_GRAINS = (
1538 (
1539 GRAIN_PARTICIPANT,
1540 SESSION_KEY,
1541 RAW_SESSION_KEY,
1542 FILE_SESSION_KEY,
1543 to_payload,
1544 from_payload,
1545 ),
1546 (
1547 "trial",
1548 TRIAL_SESSION_KEY,
1549 TRIAL_RAW_SESSION_KEY,
1550 TRIAL_FILE_SESSION_KEY,
1551 trial_to_payload,
1552 trial_from_payload,
1553 ),
1554 (
1555 "text",
1556 TEXT_SESSION_KEY,
1557 TEXT_RAW_SESSION_KEY,
1558 TEXT_FILE_SESSION_KEY,
1559 text_to_payload,
1560 text_from_payload,
1561 ),
1562)
1563_GRAIN_KEYS = {grain: (key, raw, file) for grain, key, raw, file, *_ in _GRAINS}
1566#: The payloads' row lists — `to_payload` says ``records``, the other two ``rows``.
1567_ROW_KEYS = ("records", "rows")
1570def session_payloads(session) -> dict[str, dict]:
1571 """Every attached table as its save & restore payload, keyed by grain.
1573 Each carries its frame's ``columns`` too: the cache writes its manifest with
1574 sorted keys, which would otherwise hand the rows back alphabetised and
1575 reorder the table's fields everywhere they are listed.
1576 """
1577 payloads = {}
1578 for grain, key, _raw, _file, dump, _load in _GRAINS:
1579 attached = session.get(key)
1580 payload = dump(attached)
1581 if payload is not None:
1582 payloads[grain] = {**payload, "columns": list(attached.frame.columns)}
1583 return payloads
1586def _in_column_order(payload):
1587 """``payload`` with each row's keys back in its ``columns`` order."""
1588 columns = payload.get("columns") if isinstance(payload, dict) else None
1589 if not columns:
1590 return payload
1591 ordered = dict(payload)
1592 for rows_key in _ROW_KEYS:
1593 rows = payload.get(rows_key)
1594 if isinstance(rows, list):
1595 ordered[rows_key] = [
1596 {column: row[column] for column in columns if column in row}
1597 for row in rows
1598 if isinstance(row, dict)
1599 ]
1600 return ordered
1603def session_signature(session) -> list:
1604 """A cheap content fingerprint of the attached tables.
1606 For the recovery cache's every-rerun "did anything change" check. Object
1607 identity will not do: the tables are rebuilt on every render of the Data
1608 page, and the participant one is re-joined on every run, so a new object
1609 arrives when nothing changed. The frames are small (one row per reader,
1610 trial or text), so hashing their content is cheap.
1611 """
1612 signature = []
1613 for grain, key, *_ in _GRAINS:
1614 attached = session.get(key)
1615 frame = getattr(attached, "frame", None)
1616 if not isinstance(frame, pd.DataFrame) or frame.empty:
1617 continue
1618 try:
1619 # Row hashes in row order — a sum would miss a reordered table.
1620 cells = pd.util.hash_pandas_object(frame, index=False).to_numpy().tobytes()
1621 except (TypeError, ValueError): # unhashable cells — hash their text
1622 cells = frame.to_csv(index=False).encode("utf-8")
1623 digest = hashlib.sha256(cells).hexdigest()
1624 signature.append(
1625 [
1626 grain,
1627 str(getattr(attached, "source_name", "")),
1628 list(frame.columns),
1629 digest,
1630 ]
1631 )
1632 return signature
1635def grain_keys(grain: str) -> tuple[str, str, str]:
1636 """``(table key, raw key, file key)`` in session state for ``grain``."""
1637 return _GRAIN_KEYS[grain]
1640def mark_restored(session, grain: str, attached) -> None:
1641 """Attach ``attached`` as a table with no live upload behind it.
1643 The recovery cache hands back a table the uploader never saw — see
1644 :data:`RESTORED_FILE_SIGNATURE`.
1645 """
1646 key, raw, file = _GRAIN_KEYS[grain]
1647 session[key] = attached
1648 session[raw] = attached.frame
1649 session[file] = RESTORED_FILE_SIGNATURE
1652def is_restored(session, grain: str) -> bool:
1653 """Whether ``grain``'s attached table came back without a file behind it."""
1654 return session.get(_GRAIN_KEYS[grain][2]) == RESTORED_FILE_SIGNATURE
1657def restore_payloads(session, payloads) -> int:
1658 """Re-attach the tables :func:`session_payloads` wrote; how many landed.
1660 A grain already attached in this session keeps its own table — the same
1661 "never overwrite what is already seeded" rule the rest of the restore
1662 follows — and a payload that no longer builds is skipped, not raised: a
1663 stale cache must never stop the app opening.
1664 """
1665 if not isinstance(payloads, dict):
1666 return 0
1667 restored = 0
1668 for grain, key, _raw, _file, _dump, load in _GRAINS:
1669 if session.get(key) is not None:
1670 continue
1671 try:
1672 attached = load(_in_column_order(payloads.get(grain)))
1673 except (ValueError, TypeError, KeyError):
1674 attached = None
1675 if attached is None:
1676 continue
1677 mark_restored(session, grain, attached)
1678 restored += 1
1679 return restored
1682#: DATA-47 — every dataset's tables but the selected one's, as the payloads
1683#: :func:`session_payloads` builds: ``{dataset: {grain: payload}}``. Payloads
1684#: rather than table objects so the cache can write them as they are.
1685DATASET_STORE_KEY = "_metadata_by_dataset"
1686#: Which dataset the session keys' tables belong to right now.
1687OWNER_KEY = "_metadata_owner"
1688#: Bumped on every change to the store — the cache's cheap "did it change" test,
1689#: since hashing every stored table on every rerun would not be cheap.
1690STORE_REVISION_KEY = "_metadata_store_revision"
1691#: The add-dataset wizard's dataset, before it has a name. Never cached.
1692PENDING_DATASET = "\x00pending"
1693#: Bumped by :func:`reset_uploads`; part of every metadata uploader's key.
1694UPLOAD_GENERATION_KEY = "_metadata_upload_generation"
1697def upload_key(grain: str, session=None) -> str:
1698 """``grain``'s uploader widget key, in the current upload generation.
1700 Popping an uploader's key from session state does not empty it: the browser
1701 still holds the file and sends it back on the next rerun, where it reads as
1702 a new upload — so a table attached to one dataset re-attached itself to the
1703 next one opened. Only a new key gives the browser a fresh, empty uploader,
1704 so a dataset switch moves every uploader to a new generation.
1705 """
1706 if session is None:
1707 import streamlit as st
1709 session = st.session_state
1710 generation = int(session.get(UPLOAD_GENERATION_KEY) or 0)
1711 base = f"{grain}_metadata_upload"
1712 return base if generation == 0 else f"{base}_{generation}"
1715def reset_uploads(session) -> None:
1716 """Empty every metadata uploader, in the browser too (:func:`upload_key`)."""
1717 for grain, *_ in _GRAINS:
1718 session.pop(upload_key(grain, session), None)
1719 session[UPLOAD_GENERATION_KEY] = int(session.get(UPLOAD_GENERATION_KEY) or 0) + 1
1722def _widget_keys(grain: str) -> tuple[str, ...]:
1723 """The UI state of ``grain``'s section that describes one dataset's table.
1725 The display name, the id-column and keep-fields picks, so the next
1726 dataset's table starts from its own auto-detect. The uploader is emptied
1727 separately, by :func:`reset_uploads`.
1728 """
1729 return (
1730 f"_{grain}_metadata_name",
1731 f"{grain}_metadata_id_column",
1732 f"{grain}_metadata_keep_fields",
1733 )
1736def clear_active(session) -> None:
1737 """Detach the selected dataset's tables from the session keys — all grains."""
1738 for grain, key, raw, file, *_ in _GRAINS:
1739 for name in (key, raw, file, *_widget_keys(grain)):
1740 session.pop(name, None)
1741 reset_uploads(session)
1744def _store(session) -> dict:
1745 store = session.get(DATASET_STORE_KEY)
1746 return dict(store) if isinstance(store, dict) else {}
1749def _set_store(session, store: dict) -> None:
1750 session[DATASET_STORE_KEY] = store
1751 session[STORE_REVISION_KEY] = int(session.get(STORE_REVISION_KEY) or 0) + 1
1754def stash_active(session) -> None:
1755 """File the session keys' tables under the dataset they belong to."""
1756 owner = session.get(OWNER_KEY)
1757 if owner is None:
1758 return
1759 store = _store(session)
1760 payloads = session_payloads(session)
1761 if payloads:
1762 if store.get(owner) == payloads:
1763 return # unchanged since it was restored — nothing for the cache to do
1764 store[owner] = payloads
1765 elif owner not in store:
1766 return
1767 else:
1768 store.pop(owner)
1769 _set_store(session, store)
1772def activate_dataset(session, dataset: str) -> bool:
1773 """Make ``dataset``'s tables the attached ones; whether anything moved.
1775 Called by ``app.main`` on every run with the selected dataset. When the
1776 selection changed, the outgoing dataset's tables are filed away, the session
1777 keys are cleared — widgets included, see :func:`_widget_keys` — and the
1778 incoming dataset's are restored (marked restored, since no uploader holds
1779 their file). A session whose tables have no owner yet (its first run) adopts
1780 whatever is attached for ``dataset`` rather than clearing it.
1781 """
1782 dataset = str(dataset)
1783 owner = session.get(OWNER_KEY)
1784 if owner == dataset:
1785 return False
1786 if owner is not None:
1787 stash_active(session)
1788 clear_active(session)
1789 session[OWNER_KEY] = dataset
1790 restore_payloads(session, _store(session).get(dataset))
1791 return True
1794#: CMP-8's key prefix for scanpath B's filters, and the picker's "same dataset"
1795#: answer (``compare_source.THIS_DATASET`` — not imported: `compare_source`
1796#: imports `app`, which imports this).
1797_COMPARE_PREFIX = "cmp"
1798_COMPARE_SAME_DATASET = "This dataset"
1799_BUILT_KEY = "_metadata_built_for_compare"
1802#: EXP-22 — the table each grain is named by in a title / caption pattern:
1803#: ``{trials.font_size}`` is the trial table's ``font_size``.
1804PATTERN_TABLE_NAMES = {
1805 GRAIN_PARTICIPANT: "participants",
1806 GRAIN_TRIAL: "trials",
1807 GRAIN_TEXT: "texts",
1808}
1811def pattern_rows(
1812 participant, trial, text_id=None, *, prefix: str = ""
1813) -> dict[str, dict[str, object]]:
1814 """This trial's row of every attached metadata table (EXP-22).
1816 ``{"participants": {...}, "trials": {...}, "texts": {...}}`` — only the
1817 tables that are attached, each with every registered field (a reader, trial
1818 or text the table does not mention gets ``None`` values, so the field still
1819 exists and a pattern naming it still validates). What
1820 ``export.table_pattern_fields`` turns into ``{trials.font_size}``.
1821 """
1822 rows: dict[str, dict[str, object]] = {}
1823 table = attached_for(GRAIN_PARTICIPANT, prefix)
1824 if table is not None:
1825 found = table.values_for(participant) if participant is not None else {}
1826 rows["participants"] = {name: found.get(name) for name in table.names}
1827 table = attached_for(GRAIN_TRIAL, prefix)
1828 if table is not None:
1829 found: dict = {}
1830 if trial is not None and not table.frame.empty:
1831 match = table.frame["trial_id"] == str(trial)
1832 if table.keyed_by_participant:
1833 match &= table.frame["participant_id"] == str(participant)
1834 hit = table.frame[match]
1835 if not hit.empty:
1836 found = hit.iloc[0].to_dict()
1837 rows["trials"] = {name: found.get(name) for name in table.names}
1838 table = attached_for(GRAIN_TEXT, prefix)
1839 if table is not None:
1840 found = table.values_for(text_id) if text_id is not None else {}
1841 rows["texts"] = {name: found.get(name) for name in table.names}
1842 return rows
1845def attached_for(grain: str, prefix: str = ""):
1846 """The table ``grain``'s filters under key ``prefix`` narrow by (DATA-47).
1848 The main pool's filters read the selected dataset's table. Compare mode's
1849 scanpath B (the ``cmp`` prefix) can come from another dataset, and then its
1850 filters must read *that* dataset's own table — which waits in the store —
1851 not A's. Built from the stored payload once per store revision.
1852 """
1853 try:
1854 import streamlit as st
1856 session = st.session_state
1857 live = session.get(_GRAIN_KEYS[grain][0])
1858 except Exception: # no script run context (API, CLI, plain import)
1859 return None
1860 if prefix != _COMPARE_PREFIX:
1861 return live
1862 other = session.get(COMPARE_SOURCE_STATE_KEY)
1863 if (
1864 not other
1865 or other == _COMPARE_SAME_DATASET
1866 or str(other) == session.get(OWNER_KEY)
1867 ):
1868 return live
1869 payload = (_store(session).get(str(other)) or {}).get(grain)
1870 if not isinstance(payload, dict):
1871 return None
1872 revision = session.get(STORE_REVISION_KEY)
1873 built = session.get(_BUILT_KEY)
1874 cache_key = (str(other), grain, revision)
1875 if not isinstance(built, dict) or cache_key not in built:
1876 load = next(entry[-1] for entry in _GRAINS if entry[0] == grain)
1877 try:
1878 table = load(_in_column_order(payload))
1879 except (ValueError, TypeError, KeyError):
1880 table = None
1881 kept = {
1882 k: v
1883 for k, v in (built or {}).items()
1884 if isinstance(k, tuple) and k[-1] == revision
1885 }
1886 session[_BUILT_KEY] = built = {**kept, cache_key: table}
1887 return built[cache_key]
1890def begin_pending_dataset(session) -> None:
1891 """Start the add-dataset wizard's dataset with no tables of its own."""
1892 store = _store(session)
1893 if PENDING_DATASET in store:
1894 store.pop(PENDING_DATASET)
1895 _set_store(session, store)
1898def adopt_pending_dataset(session, dataset: str) -> None:
1899 """✅ Add dataset: the wizard's tables become ``dataset``'s.
1901 The session keys already hold them (the deferred join has just attached
1902 them), so this only renames who owns them — the next run's
1903 :func:`activate_dataset` then sees nothing to swap.
1904 """
1905 begin_pending_dataset(session)
1906 session[OWNER_KEY] = str(dataset)
1909def forget_dataset(session, dataset: str) -> None:
1910 """A removed dataset's tables go with it."""
1911 dataset = str(dataset)
1912 store = _store(session)
1913 if dataset in store:
1914 store.pop(dataset)
1915 _set_store(session, store)
1916 if session.get(OWNER_KEY) == dataset:
1917 clear_active(session)
1918 session.pop(OWNER_KEY, None)
1921def rename_dataset(session, old: str, new: str) -> None:
1922 """A renamed dataset keeps its tables."""
1923 old, new = str(old), str(new)
1924 store = _store(session)
1925 if old in store:
1926 _set_store(session, {(new if k == old else k): v for k, v in store.items()})
1927 if session.get(OWNER_KEY) == old:
1928 session[OWNER_KEY] = new
1931def dataset_payloads(session) -> dict[str, dict]:
1932 """Every dataset's tables, the selected one's live: ``{dataset: {grain: …}}``.
1934 What the recovery cache writes. The add-dataset wizard's unnamed dataset is
1935 left out — it is not a dataset yet, and a restart discards the wizard.
1936 """
1937 store = _store(session)
1938 owner = session.get(OWNER_KEY)
1939 if owner is not None:
1940 live = session_payloads(session)
1941 if live:
1942 store[owner] = live
1943 else:
1944 store.pop(owner, None)
1945 store.pop(PENDING_DATASET, None)
1946 return {name: payloads for name, payloads in store.items() if payloads}
1949def store_signature(session) -> list:
1950 """A cheap fingerprint of every dataset's tables, for the cache (DATA-47).
1952 The live tables by content (:func:`session_signature` — they are rebuilt on
1953 every render), the rest by the store's revision counter. Empty when nothing
1954 is attached anywhere, which is what tells the cache to delete its file.
1955 """
1956 live = session_signature(session)
1957 # Deliberately not `dataset_payloads`, which serializes the live tables: this
1958 # runs on every rerun, and the live half is already covered by `live`.
1959 owner = session.get(OWNER_KEY)
1960 stored = any(
1961 tables
1962 for name, tables in _store(session).items()
1963 if name not in (owner, PENDING_DATASET)
1964 )
1965 if not live and not stored:
1966 return []
1967 return [
1968 ["store", int(session.get(STORE_REVISION_KEY) or 0)],
1969 ["owner", str(session.get(OWNER_KEY))],
1970 *live,
1971 ]
1974def restore_dataset_payloads(session, payloads) -> int:
1975 """Put the tables :func:`dataset_payloads` wrote back in the store; how many.
1977 A dataset this session already holds tables for keeps its own. The selected
1978 dataset's go straight onto the session keys, grain by grain — one already
1979 attached is kept, like the rest of the restore's ``setdefault`` — and the
1980 others wait in the store for :func:`activate_dataset`.
1981 """
1982 datasets = payloads.get("datasets") if isinstance(payloads, dict) else None
1983 if not isinstance(datasets, dict):
1984 return 0
1985 store = _store(session)
1986 owner = session.get(OWNER_KEY)
1987 restored = 0
1988 changed = False
1989 for name, tables in datasets.items():
1990 name = str(name)
1991 if not isinstance(tables, dict) or name == PENDING_DATASET or name in store:
1992 continue
1993 store[name] = tables
1994 changed = True
1995 if name == owner:
1996 restored += restore_payloads(session, tables)
1997 else:
1998 restored += sum(
1999 1 for grain, *_ in _GRAINS if isinstance(tables.get(grain), dict)
2000 )
2001 if changed:
2002 _set_store(session, store)
2003 return restored