Coverage for scanpath_studio/metadata.py: 93%

826 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Keyed, entity-level metadata tables (DATA-20). 

2 

3Milestone 1 is **participant grain**: a separate table whose rows are readers 

4and whose columns (``native_language``, ``age``, ``comprehension_score``, …) 

5should behave "as if they were fields in the data" — filterable, chip-able, 

6sortable, inspectable, exportable — without being any part of the recorded eye 

7movements. 

8 

9Two rules shape everything here. 

10 

11**The table stays separate.** It is never broadcast across every word/fixation 

12row: a 40-row participant table joined onto three million fixations costs 

13memory and buys nothing, and it would make a *reader* attribute look like a 

14per-fixation measurement. Instead the frame is kept as-is and consumed at three 

15narrow boundaries: 

16 

17* *filtering* — a participant-grain constraint is a **participant** constraint, 

18 so :func:`participants_matching` turns a selection into the set of ids the 

19 existing participant filter already knows how to apply. No join at all. 

20* *projection* — :func:`project` left-joins chosen columns onto a **small** 

21 frame (the per-trial ``combos`` table, a group-by result), which is where 

22 sorting, grouping and chips read from. 

23* *display / export* — the frame itself, shown and written as its own table. 

24 

25**Nothing is silently collapsed.** Duplicate participant rows that disagree are 

26not resolved by taking the first one: the id is reported as *conflicting* and 

27contributes no value, so a downstream field reads as missing rather than as an 

28arbitrary winner. Unmatched ids are reported on both sides — rows describing 

29readers who are not in the data, and readers in the data with no row. 

30 

31**Milestone 2 is trial grain (DATA-29)**: the same idea one level down — a 

32table whose rows are *readings*. It reuses everything that is about validating a 

33keyed table (the dtype classification, the field registry, the join report, the 

34"conflicting rows are dropped, not resolved" rule) and differs only where the 

35grain genuinely differs: 

36 

37* the **key** is the user's call — a trial id alone, or a reader **and** a trial 

38 id, because a repeated reading is a different trial for the same reader only 

39 in corpora that record it that way; 

40* **filtering** narrows to a set of ``(participant_id, trial_id)`` keys, applied 

41 by ``data.filter_to_keys`` — there is no participant-constraint indirection to 

42 mirror, because a trial constraint already *is* the grain the pool is keyed on. 

43 

44Later grains (stimulus, screen, word, fixation) add rows to the same registry; 

45:class:`MetadataField` already carries ``grain``. 

46""" 

47 

48from __future__ import annotations 

49 

50import hashlib 

51from collections.abc import Iterable, Mapping, Sequence 

52from dataclasses import dataclass, replace 

53 

54import numpy as np 

55import pandas as pd 

56 

57from . import data as _data 

58from .data import ( 

59 composite_respelling_map, 

60 stable_id, 

61 trial_id_series, 

62 trial_mapping_columns, 

63 zero_padding_map, 

64) 

65from .session_keys import COMPARE_SOURCE_STATE_KEY 

66 

67 

68def _with_data_candidates(data_candidates: list[str], *extras: str) -> tuple[str, ...]: 

69 """``data``'s own candidate list, in its order, then the metadata-only 

70 spellings it does not already hold (compared case-insensitively, as the 

71 ``infer_*_id_column`` lookups compare). 

72 

73 DATA-44: the metadata lists used to be hand-copied twins of ``data``'s and 

74 drifted — the text list lost ``unique_paragraph_id``, the demo corpus's own 

75 text id — so a table exported beside the data needed a manual pick. Deriving 

76 them means a name ``data`` learns is a name a metadata table is keyed by.""" 

77 seen: set[str] = set() 

78 out: list[str] = [] 

79 for name in (*data_candidates, *extras): 

80 if name.lower() not in seen: 

81 seen.add(name.lower()) 

82 out.append(name) 

83 return tuple(out) 

84 

85 

86# Source columns that plausibly hold the reader id, most explicit first: the 

87# names `data.PARTICIPANT_CANDIDATES` maps the reader from, so a metadata file 

88# exported beside the data usually needs no picking, plus a few spellings only a 

89# hand-made readers table tends to use. First hit wins, and the user can always 

90# override the guess in the UI. 

91PARTICIPANT_ID_CANDIDATES: tuple[str, ...] = _with_data_candidates( 

92 _data.PARTICIPANT_CANDIDATES, "participant", "subject", "reader", "pid" 

93) 

94 

95# Grain of a field — the entity one row describes. PARTICIPANT (DATA-20), 

96# TRIAL (DATA-29) and TEXT are ingested; the rest are named so the registry's 

97# shape is settled. 

98GRAIN_PARTICIPANT = "participant" 

99GRAIN_TRIAL = "trial" 

100GRAIN_TEXT = "text" 

101 

102# Source columns that plausibly hold the trial id — the trial-grain twin of 

103# PARTICIPANT_ID_CANDIDATES, derived from `data.TRIAL_CANDIDATES` the same way. 

104TRIAL_ID_CANDIDATES: tuple[str, ...] = _with_data_candidates( 

105 _data.TRIAL_CANDIDATES, "item_id" 

106) 

107 

108# The text-grain twin, derived from `data.TEXT_ID_CANDIDATES`. 

109TEXT_ID_CANDIDATES: tuple[str, ...] = _with_data_candidates( 

110 _data.TEXT_ID_CANDIDATES, "text", "item_id", "stimulus_id" 

111) 

112 

113# Loader bookkeeping, never user metadata: `data.read_tables` tags each row with 

114# the file it came from, which would otherwise be registered as a field called 

115# "Source file" and offered as a filter and a chip. Excluded here, in the one 

116# place every ingestion route passes through, rather than at each caller. 

117_BOOKKEEPING_COLUMNS = frozenset({"source_file"}) 

118 

119# Session state: the validated table, and the raw frame it was built from (kept 

120# so a different id column can be picked without re-uploading the file). Both 

121# are plain session state rather than widget keys — they are not wire format, 

122# and `session_keys.py` deliberately does not pin them. 

123SESSION_KEY = "_participant_metadata" 

124RAW_SESSION_KEY = "_participant_metadata_raw" 

125FILE_SESSION_KEY = "_participant_metadata_file" 

126 

127# DATA-29 — the same three, for the trial table. Separate keys rather than one 

128# keyed-by-grain dict: the two tables are attached, replaced and cleared 

129# independently, and every consumer wants one of them specifically. 

130TRIAL_SESSION_KEY = "_trial_metadata" 

131TRIAL_RAW_SESSION_KEY = "_trial_metadata_raw" 

132TRIAL_FILE_SESSION_KEY = "_trial_metadata_file" 

133 

134# The same three, for the text table (one row per text_id) — the third grain. 

135TEXT_SESSION_KEY = "_text_metadata" 

136TEXT_RAW_SESSION_KEY = "_text_metadata_raw" 

137TEXT_FILE_SESSION_KEY = "_text_metadata_file" 

138 

139_DTYPE_CATEGORICAL = "categorical" 

140_DTYPE_NUMERIC = "numeric" 

141_DTYPE_BOOLEAN = "boolean" 

142 

143 

144@dataclass(frozen=True) 

145class MetadataField: 

146 """One registered column, with everything a consumer needs to place it.""" 

147 

148 name: str 

149 label: str 

150 grain: str 

151 dtype: str 

152 source: str 

153 n_unique: int 

154 n_missing: int 

155 

156 @property 

157 def is_numeric(self) -> bool: 

158 return self.dtype == _DTYPE_NUMERIC 

159 

160 @property 

161 def is_categorical(self) -> bool: 

162 return self.dtype in (_DTYPE_CATEGORICAL, _DTYPE_BOOLEAN) 

163 

164 

165@dataclass(frozen=True) 

166class JoinReport: 

167 """What happened when the table met the participants actually loaded. 

168 

169 Every count is a list of ids rather than a number so the UI can name them — 

170 "3 unmatched" is not actionable, "``p07``, ``p12``, ``p31``" is. 

171 """ 

172 

173 matched: tuple[str, ...] = () 

174 only_in_table: tuple[str, ...] = () 

175 only_in_data: tuple[str, ...] = () 

176 duplicated: tuple[str, ...] = () 

177 conflicting: tuple[str, ...] = () 

178 #: How many rows of the file were folded together because they repeated a 

179 #: key without disagreeing (each field takes the one value its rows hold). 

180 combined_rows: int = 0 

181 

182 @property 

183 def is_clean(self) -> bool: 

184 return not ( 

185 self.only_in_table 

186 or self.only_in_data 

187 or self.duplicated 

188 or self.conflicting 

189 ) 

190 

191 

192@dataclass(frozen=True) 

193class ParticipantMetadata: 

194 """A validated participant table plus its field registry. 

195 

196 ``frame`` is indexed by nothing in particular but always carries a string 

197 ``participant_id`` column; conflicting ids have been dropped from it (and 

198 named in :attr:`report`), so a lookup either finds one unambiguous row or 

199 finds none. 

200 """ 

201 

202 frame: pd.DataFrame 

203 fields: tuple[MetadataField, ...] 

204 source_name: str 

205 id_column: str 

206 report: JoinReport = JoinReport() 

207 

208 @property 

209 def names(self) -> tuple[str, ...]: 

210 return tuple(field.name for field in self.fields) 

211 

212 def field(self, name: str) -> MetadataField | None: 

213 for candidate in self.fields: 

214 if candidate.name == name: 

215 return candidate 

216 return None 

217 

218 def values_for(self, participant_id) -> dict[str, object]: 

219 """Every registered value for one reader (missing ids give ``{}``).""" 

220 if self.frame.empty: 

221 return {} 

222 match = self.frame[self.frame["participant_id"] == str(participant_id)] 

223 if match.empty: 

224 return {} 

225 row = match.iloc[0] 

226 return {name: row[name] for name in self.names if name in match.columns} 

227 

228 @property 

229 def joined_frame(self) -> pd.DataFrame: 

230 """Only the rows describing readers that are actually loaded. 

231 

232 What the *controls* must be built from. Offering a value that belongs to 

233 a reader the report has just called "not loaded — ignored" gives the 

234 user a filter that can only ever empty the pool, and stretches a numeric 

235 slider to a bound nobody in the data has. With no participant list to 

236 join against (``participants=None``), the report matches everything and 

237 this is the whole frame. 

238 """ 

239 if self.frame.empty: 

240 return self.frame 

241 return self.frame[self.frame["participant_id"].isin(set(self.report.matched))] 

242 

243 def series(self, name: str) -> pd.Series: 

244 """``participant_id`` → value for one field, for projection/lookup.""" 

245 if self.frame.empty or name not in self.frame.columns: 

246 return pd.Series(dtype="object") 

247 return self.frame.set_index("participant_id")[name] 

248 

249 

250@dataclass(frozen=True) 

251class TrialMetadata: 

252 """A validated trial table plus its field registry (DATA-29). 

253 

254 ``frame`` always carries a string ``trial_id`` column, and a string 

255 ``participant_id`` column as well when the table is keyed by both. Rows 

256 whose key repeats *with different values* have been dropped and named in 

257 :attr:`report`, so a lookup either finds one unambiguous row or none — the 

258 same rule the participant table follows. 

259 

260 ``keyed_by_participant`` is the user's answer to the question DATA-29 opened 

261 with. It is not inferred: a corpus where every reader reads every text can 

262 key by trial id alone and mean it, and one with repeated readings cannot, 

263 and nothing in the file itself says which world you are in. 

264 """ 

265 

266 frame: pd.DataFrame 

267 fields: tuple[MetadataField, ...] 

268 source_name: str 

269 trial_column: str 

270 participant_column: str | None = None 

271 report: JoinReport = JoinReport() 

272 

273 @property 

274 def keyed_by_participant(self) -> bool: 

275 return bool(self.participant_column) 

276 

277 @property 

278 def key_columns(self) -> tuple[str, ...]: 

279 return ( 

280 ("participant_id", "trial_id") 

281 if self.keyed_by_participant 

282 else ("trial_id",) 

283 ) 

284 

285 @property 

286 def names(self) -> tuple[str, ...]: 

287 return tuple(field.name for field in self.fields) 

288 

289 def field(self, name: str) -> MetadataField | None: 

290 for candidate in self.fields: 

291 if candidate.name == name: 

292 return candidate 

293 return None 

294 

295 def key_series(self) -> pd.Series: 

296 """The frame's own keys, as the string tuples the reports speak in.""" 

297 if self.frame.empty: 

298 return pd.Series(dtype="object") 

299 if self.keyed_by_participant: 

300 return pd.Series( 

301 list( 

302 zip( 

303 self.frame["participant_id"].astype(str), 

304 self.frame["trial_id"].astype(str), 

305 ) 

306 ), 

307 index=self.frame.index, 

308 ) 

309 return self.frame["trial_id"].astype(str) 

310 

311 @property 

312 def joined_frame(self) -> pd.DataFrame: 

313 """Only the rows describing trials that are actually loaded. 

314 

315 What the *controls* are built from, for `ParticipantMetadata`'s reason: 

316 offering a value that belongs to a trial the report has just called "not 

317 loaded — ignored" gives the user a filter that can only empty the pool. 

318 """ 

319 if self.frame.empty: 

320 return self.frame 

321 return self.frame[self.key_series().isin(set(self.report.matched))] 

322 

323 def values_for(self, participant_id, trial_id) -> dict[str, object]: 

324 """Every registered value for one reading (an unknown key gives ``{}``).""" 

325 if self.frame.empty: 

326 return {} 

327 match = self.frame[self.frame["trial_id"] == str(trial_id)] 

328 if self.keyed_by_participant: 

329 match = match[match["participant_id"] == str(participant_id)] 

330 if match.empty: 

331 return {} 

332 row = match.iloc[0] 

333 return {name: row[name] for name in self.names if name in match.columns} 

334 

335 

336@dataclass(frozen=True) 

337class TextMetadata: 

338 """A validated text table plus its field registry — the third grain. 

339 

340 ``frame`` always carries a string ``text_id`` column; conflicting ids have 

341 been dropped from it (and named in :attr:`report`), the same rule 

342 :class:`ParticipantMetadata`/:class:`TrialMetadata` follow. Flat grain — 

343 one row per text, joined the way :class:`ParticipantMetadata` joins by 

344 reader, never :class:`TrialMetadata`'s participant-pairing option: a text 

345 is a stimulus, not something one reader owns. 

346 """ 

347 

348 frame: pd.DataFrame 

349 fields: tuple[MetadataField, ...] 

350 source_name: str 

351 text_column: str 

352 report: JoinReport = JoinReport() 

353 

354 @property 

355 def names(self) -> tuple[str, ...]: 

356 return tuple(field.name for field in self.fields) 

357 

358 def field(self, name: str) -> MetadataField | None: 

359 for candidate in self.fields: 

360 if candidate.name == name: 

361 return candidate 

362 return None 

363 

364 def values_for(self, text_id) -> dict[str, object]: 

365 """Every registered value for one text (an unknown id gives ``{}``).""" 

366 if self.frame.empty: 

367 return {} 

368 match = self.frame[self.frame["text_id"] == str(text_id)] 

369 if match.empty: 

370 return {} 

371 row = match.iloc[0] 

372 return {name: row[name] for name in self.names if name in match.columns} 

373 

374 @property 

375 def joined_frame(self) -> pd.DataFrame: 

376 """Only the rows describing texts that are actually loaded. 

377 

378 Same reasoning as :attr:`ParticipantMetadata.joined_frame` — the 

379 controls must be built from what is on screen, not from every text 

380 the table happens to mention. 

381 """ 

382 if self.frame.empty: 

383 return self.frame 

384 return self.frame[self.frame["text_id"].isin(set(self.report.matched))] 

385 

386 def series(self, name: str) -> pd.Series: 

387 """``text_id`` → value for one field, for projection/lookup.""" 

388 if self.frame.empty or name not in self.frame.columns: 

389 return pd.Series(dtype="object") 

390 return self.frame.set_index("text_id")[name] 

391 

392 

393def _rows_with_ids(frame: pd.DataFrame, columns) -> pd.DataFrame: 

394 """A copy of ``frame`` without the rows that have no value in an id column. 

395 

396 A blank row — the one Excel leaves at the end of a sheet — became a phantom 

397 reader named "nan": under pandas 3 a missing id stays NaN through 

398 ``stable_id``, and the ``!= ""`` test that used to drop it let NaN 

399 through (BUG-60). A composite id with a missing part raised in the join 

400 instead. Such a row describes no one, so it goes. 

401 """ 

402 ids = frame[list(columns)] 

403 missing = ids.isna() | ids.apply(lambda c: c.astype(str).str.strip() == "") 

404 return frame.loc[~missing.any(axis=1)].copy() 

405 

406 

407def _merge_duplicates(work: pd.DataFrame, key: pd.Series, value_columns) -> tuple: 

408 """Fold the rows that repeat a key, unless they disagree. 

409 

410 Returns ``(work, key, duplicated, conflicting, combined_rows)``: ``work`` 

411 and ``key`` with one row per key, the keys that repeated, the keys whose 

412 rows disagree (dropped whole, as every grain here has always done), and how 

413 many rows were folded together. 

414 

415 Rows *disagree* when a field holds two different non-missing values. Rows 

416 that do not are one record written twice, and each field takes the one 

417 value they hold — two ``p1`` rows, one with a language and one with an age, 

418 become one reader with both. Keeping the first row instead lost the second 

419 row's values, and which ones depended on the file's row order. 

420 """ 

421 repeated = key.duplicated(keep=False) 

422 if not repeated.any(): 

423 return work, key, (), set(), 0 

424 duplicated = tuple(sorted(set(key[repeated]), key=str)) 

425 columns = [column for column in value_columns if column in work.columns] 

426 conflicting: set = set() 

427 folded: list = [] 

428 combined = 0 

429 # A positional index, so writing the folded values into a group's first row 

430 # cannot also land on another row a user's frame gave the same label. 

431 work, key = work.reset_index(drop=True), key.reset_index(drop=True) 

432 repeated = repeated.reset_index(drop=True) 

433 for group_key, rows in work[repeated].groupby(key[repeated], sort=False): 

434 if any(rows[column].dropna().nunique() > 1 for column in columns): 

435 conflicting.add(group_key) 

436 continue 

437 first = rows.index[0] 

438 for column in columns: 

439 present = rows[column].dropna() 

440 if not present.empty: 

441 work.at[first, column] = present.iloc[0] 

442 folded.extend(rows.index[1:]) 

443 combined += len(rows) 

444 keep = ~key.isin(conflicting) & ~work.index.isin(folded) 

445 return work[keep], key[keep], duplicated, conflicting, combined 

446 

447 

448def _range_mask( 

449 frame: pd.DataFrame, 

450 ranges: Mapping[str, tuple[float, float]], 

451 keep_unknown: Mapping[str, bool] | None, 

452) -> pd.Series: 

453 """Rows inside every range in ``ranges`` (inclusive). 

454 

455 A row with no value for a ranged field is kept — a range narrows, it does 

456 not exclude the unmeasured (UX-49) — unless ``keep_unknown`` maps that 

457 field to ``False``: the researcher's explicit "only records with a 

458 measured value". 

459 """ 

460 mask = pd.Series(True, index=frame.index) 

461 for name, (low, high) in ranges.items(): 

462 numeric = pd.to_numeric(frame[name], errors="coerce") 

463 inside = numeric.between(low, high) 

464 if (keep_unknown or {}).get(name, True) is not False: 

465 inside |= numeric.isna() 

466 mask &= inside 

467 return mask 

468 

469 

470def _keeps_unlisted( 

471 ranges: Mapping[str, tuple[float, float]], 

472 keep_unknown: Mapping[str, bool] | None, 

473) -> bool: 

474 """Whether a record the table has **no row** for survives ``ranges``. 

475 

476 It has no value for any field, so it is unknown for every range: kept 

477 while every active range keeps its unknowns, left out once one does not. 

478 """ 

479 return all((keep_unknown or {}).get(name, True) is not False for name in ranges) 

480 

481 

482def numeric_extent(metadata, name: str) -> tuple[float, float] | None: 

483 """``(min, max)`` of a numeric field over the loaded records, **equal ends 

484 included** — or ``None`` when no loaded record has a value. 

485 

486 :func:`bounds_for` and its siblings answer "is there a range to slide 

487 over?" and so return ``None`` for a constant field. This one answers "what 

488 values are there?", which a constant field still has: the filter panel 

489 shows it as a fixed value, and its *Keep unknown values* choice can still 

490 leave out the records that have none. Works on any of the three tables. 

491 """ 

492 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

493 return None 

494 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna() 

495 if numeric.empty: 

496 return None 

497 return float(numeric.min()), float(numeric.max()) 

498 

499 

500def unknown_count(metadata, name: str, keys: Iterable | None = None) -> int: 

501 """How many loaded records have no value for ``name`` — the unknowns a 

502 range keeps or, with *Keep unknown values* off, leaves out. 

503 

504 Counted in the unit the filter keeps or drops: readers for the participant 

505 table, texts for the text table, and for the trial table **readings** 

506 (``(participant_id, trial_id)`` pairs) when ``keys`` — the loaded pool's 

507 pairs — is given, the table's own keys otherwise. A record with no row at 

508 all is unknown too. 

509 """ 

510 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

511 return 0 

512 if isinstance(metadata, TrialMetadata) and keys is not None: 

513 loaded = {tuple(str(part) for part in key) for key in keys} 

514 values = pd.to_numeric(metadata.frame[name], errors="coerce") 

515 known = set(metadata.key_series()[values.notna()]) 

516 if metadata.keyed_by_participant: 

517 return sum(1 for key in loaded if key not in known) 

518 return sum(1 for key in loaded if len(key) < 2 or key[1] not in known) 

519 values = pd.to_numeric(metadata.joined_frame[name], errors="coerce") 

520 return int(values.isna().sum()) + len(metadata.report.only_in_data) 

521 

522 

523def active_trials() -> TrialMetadata | None: 

524 """The trial table attached to this session, or ``None`` (DATA-29).""" 

525 try: 

526 import streamlit as st 

527 

528 return st.session_state.get(TRIAL_SESSION_KEY) 

529 except Exception: # no script run context (API, CLI, plain import) 

530 return None 

531 

532 

533def trial_keys(combos: pd.DataFrame | None) -> set: 

534 """The ``(participant_id, trial_id)`` pairs the loaded data actually has.""" 

535 if combos is None or combos.empty: 

536 return set() 

537 if not {"participant_id", "trial_id"} <= set(combos.columns): 

538 return set() 

539 pairs = combos[["participant_id", "trial_id"]].astype(str).drop_duplicates() 

540 return set(map(tuple, pairs.to_numpy())) 

541 

542 

543def infer_trial_id_column(frame: pd.DataFrame) -> str | None: 

544 """First plausible trial-id column, or ``None`` — the UI's initial guess.""" 

545 if frame is None or frame.empty: 

546 return None 

547 lookup = {str(column).lower(): str(column) for column in frame.columns} 

548 for candidate in TRIAL_ID_CANDIDATES: 

549 hit = lookup.get(candidate.lower()) 

550 if hit is not None: 

551 return hit 

552 return None 

553 

554 

555def _trial_column_label(trial_column) -> str: 

556 """Display form of a (possibly composite) trial-column mapping — joined 

557 with " + ", matching how the wizard spells a composite trial id back out.""" 

558 return " + ".join(trial_mapping_columns(trial_column)) 

559 

560 

561def build_trial_metadata( 

562 frame: pd.DataFrame, 

563 trial_column: str | list[str], 

564 participant_column: str | None = None, 

565 *, 

566 source_name: str = "trial metadata", 

567 keys: Iterable | None = None, 

568) -> TrialMetadata: 

569 """Validate a raw trial table into a registry + clean frame (DATA-29). 

570 

571 ``trial_column`` is a single column name, or **several** to build a unique 

572 trial id on the fly (joined with ``_``, like the Trial ID mapping the 

573 uploaded data itself uses — see :func:`data.trial_id_series`) — for a 

574 table whose own trial id needs the same composite key the data does. 

575 

576 The participant half of the key is **optional and explicit**: pass 

577 ``participant_column`` to key by reader *and* trial. ``keys`` is the set of 

578 ``(participant_id, trial_id)`` pairs present in the loaded data, which fills 

579 in the two "unmatched" halves of the report — and, when the table is keyed by 

580 trial alone, is collapsed to trial ids first so a table that legitimately 

581 describes one reading per text is not reported as missing every reader. 

582 

583 Mirrors :func:`build_participant_metadata` deliberately, including the rule 

584 that duplicate rows are only a problem when they *disagree*. 

585 """ 

586 trial_cols = trial_mapping_columns(trial_column) 

587 label = _trial_column_label(trial_column) 

588 empty = TrialMetadata( 

589 pd.DataFrame(columns=["trial_id"]), 

590 (), 

591 source_name, 

592 label, 

593 str(participant_column) if participant_column else None, 

594 ) 

595 if ( 

596 frame is None 

597 or frame.empty 

598 or not trial_cols 

599 or any(c not in frame.columns for c in trial_cols) 

600 ): 

601 return empty 

602 if participant_column and participant_column not in frame.columns: 

603 participant_column = None 

604 

605 work = _rows_with_ids( 

606 frame, [*trial_cols, *([participant_column] if participant_column else [])] 

607 ) 

608 # `trial_id_series` — not a plain `.astype(str)` — so this table's own 

609 # trial id is spelled the same way `data.normalize_*` spells the app's: a 

610 # blank cell anywhere else in *this* file's trial-id column is enough to 

611 # read it as floats ("101.0") against the data's "101", and the join below 

612 # would silently match nothing (DATA-29's "no reading matched" is exactly 

613 # this) — and a composite id is built the identical way (`compose_id`, 

614 # each part through `stable_id` first). 

615 work["trial_id"] = trial_id_series(work, trial_column) 

616 if participant_column: 

617 work["participant_id"] = stable_id(work[participant_column]) 

618 reserved = { 

619 *trial_cols, 

620 str(participant_column) if participant_column else "", 

621 "trial_id", 

622 "participant_id", 

623 *_BOOKKEEPING_COLUMNS, 

624 } 

625 value_columns = [ 

626 str(column) for column in frame.columns if str(column) not in reserved 

627 ] 

628 

629 key_frame = ( 

630 pd.Series(list(zip(work["participant_id"], work["trial_id"])), index=work.index) 

631 if participant_column 

632 else work["trial_id"] 

633 ) 

634 work, key_frame, duplicated, conflicting_set, combined = _merge_duplicates( 

635 work, key_frame, value_columns 

636 ) 

637 

638 clean = pd.DataFrame({"trial_id": work["trial_id"].to_numpy()}) 

639 if participant_column: 

640 clean.insert(0, "participant_id", work["participant_id"].to_numpy()) 

641 fields: list[MetadataField] = [] 

642 for column in value_columns: 

643 dtype = _classify(work[column]) 

644 values = _coerce(work[column], dtype) 

645 clean[column] = values.to_numpy() 

646 fields.append( 

647 MetadataField( 

648 name=column, 

649 label=field_label(column), 

650 grain=GRAIN_TRIAL, 

651 dtype=dtype, 

652 source=source_name, 

653 n_unique=int(values.dropna().nunique()), 

654 n_missing=int(values.isna().sum()), 

655 ) 

656 ) 

657 

658 metadata = TrialMetadata( 

659 clean, 

660 tuple(fields), 

661 source_name, 

662 label, 

663 str(participant_column) if participant_column else None, 

664 JoinReport( 

665 matched=tuple(sorted(set(key_frame), key=str)), 

666 duplicated=tuple(sorted(duplicated, key=str)), 

667 conflicting=tuple(sorted(conflicting_set, key=str)), 

668 combined_rows=combined, 

669 ), 

670 ) 

671 if keys is None: 

672 return metadata 

673 return rejoin_trials(metadata, keys) 

674 

675 

676def rejoin_trials(metadata: TrialMetadata, keys: Iterable) -> TrialMetadata: 

677 """Recompute the join report against the trials actually loaded (DATA-29). 

678 

679 ``keys`` are ``(participant_id, trial_id)`` pairs; a table keyed by trial 

680 alone is compared on the trial half, so "this file describes texts, not 

681 readings" is a supported answer rather than a report full of misses. 

682 """ 

683 data_keys = {tuple(str(part) for part in key) for key in keys} 

684 if not metadata.keyed_by_participant: 

685 data_keys = {key[1] for key in data_keys if len(key) > 1} 

686 if not metadata.frame.empty: 

687 data_trials = { 

688 key[1] if isinstance(key, tuple) else key 

689 for key in data_keys 

690 if not isinstance(key, tuple) or len(key) > 1 

691 } 

692 respelled = _respell_ids(metadata.frame["trial_id"], data_trials) 

693 if not respelled.equals(metadata.frame["trial_id"]): 

694 metadata = replace( 

695 metadata, frame=metadata.frame.assign(trial_id=respelled) 

696 ) 

697 table_keys = set(metadata.key_series()) | set(metadata.report.conflicting) 

698 usable = set(metadata.key_series()) 

699 return TrialMetadata( 

700 metadata.frame, 

701 metadata.fields, 

702 metadata.source_name, 

703 metadata.trial_column, 

704 metadata.participant_column, 

705 JoinReport( 

706 matched=tuple(sorted(usable & data_keys, key=str)), 

707 only_in_table=tuple(sorted(table_keys - data_keys, key=str)), 

708 only_in_data=tuple(sorted(data_keys - table_keys, key=str)), 

709 duplicated=metadata.report.duplicated, 

710 conflicting=metadata.report.conflicting, 

711 combined_rows=metadata.report.combined_rows, 

712 ), 

713 ) 

714 

715 

716def trials_matching( 

717 metadata: TrialMetadata | None, 

718 selections: dict[str, Sequence] | None = None, 

719 ranges: dict[str, tuple[float, float]] | None = None, 

720 *, 

721 keys: Iterable | None = None, 

722 keep_unknown: Mapping[str, bool] | None = None, 

723) -> set | None: 

724 """``(participant_id, trial_id)`` keys satisfying every trial constraint. 

725 

726 ``None`` means "no constraint" — the same contract as 

727 :func:`participants_matching`, and for the same reason: an empty selection 

728 must not narrow the pool to the trials the table happens to list. 

729 

730 ``keys`` are the loaded trials, needed for two things a trial-grain table 

731 cannot do without: expanding a trial-id-keyed table back to the readings 

732 that share that trial id, and keeping the trials the table never mentions 

733 when the only constraint is a numeric range (``data.filter_trials``' rule 

734 that a range narrows rather than excludes the unmeasured — UX-49). 

735 

736 ``keep_unknown`` maps a ranged field to ``False`` to leave its unknowns 

737 out instead: a reading whose value is missing, and one with no row at all 

738 (see :func:`_range_mask`). 

739 """ 

740 if metadata is None or metadata.frame.empty: 

741 return None 

742 active_selections = { 

743 name: list(values) 

744 for name, values in (selections or {}).items() 

745 if values and name in metadata.frame.columns 

746 } 

747 active_ranges = { 

748 name: bounds 

749 for name, bounds in (ranges or {}).items() 

750 if bounds and name in metadata.frame.columns 

751 } 

752 if not active_selections and not active_ranges: 

753 return None 

754 

755 frame = metadata.frame 

756 mask = pd.Series(True, index=frame.index) 

757 for name, values in active_selections.items(): 

758 allowed = {str(value) for value in values} 

759 mask &= frame[name].astype(str).isin(allowed) 

760 mask &= _range_mask(frame, active_ranges, keep_unknown) 

761 matching = set(metadata.key_series()[mask]) 

762 

763 loaded = {tuple(str(part) for part in key) for key in (keys or ())} 

764 if metadata.keyed_by_participant: 

765 result = {key for key in loaded if key in matching} if loaded else set(matching) 

766 else: 

767 # Trial-id grain describes every reading of that trial. 

768 result = {key for key in loaded if key[1] in matching} 

769 if ( 

770 not active_selections 

771 and loaded 

772 and _keeps_unlisted(active_ranges, keep_unknown) 

773 ): 

774 # Range-only narrowing keeps the unmeasured, including a reading with no 

775 # row at all — the participant table's rule, one grain down. "Has a 

776 # row" is read from the whole table, never from the rows that passed 

777 # the range: a reading described *outside* the range is described, and 

778 # treating it as unlisted added every one of them back. 

779 listed = set(metadata.key_series()) 

780 described = ( 

781 listed 

782 if metadata.keyed_by_participant 

783 else {key for key in loaded if key[1] in listed} 

784 ) 

785 result |= {key for key in loaded if key not in described} 

786 return result 

787 

788 

789def project_trials( 

790 metadata: TrialMetadata | None, 

791 frame: pd.DataFrame, 

792 columns: Iterable[str] | None = None, 

793) -> pd.DataFrame: 

794 """Left-join chosen trial-metadata columns onto a **small** trial-keyed frame. 

795 

796 For ``combos`` (one row per trial) — never for words or fixations, which is 

797 the rule the participant table follows for the same reason. Columns already 

798 on ``frame`` win, so a recorded column is never shadowed. 

799 """ 

800 if ( 

801 metadata is None 

802 or metadata.frame.empty 

803 or frame is None 

804 or frame.empty 

805 or "trial_id" not in frame.columns 

806 ): 

807 return frame 

808 if metadata.keyed_by_participant and "participant_id" not in frame.columns: 

809 return frame 

810 wanted = [ 

811 name 

812 for name in (list(columns) if columns is not None else list(metadata.names)) 

813 if name in metadata.frame.columns and name not in frame.columns 

814 ] 

815 if not wanted: 

816 return frame 

817 out = frame.copy() 

818 if metadata.keyed_by_participant: 

819 keys = pd.Series( 

820 list(zip(out["participant_id"].astype(str), out["trial_id"].astype(str))), 

821 index=out.index, 

822 ) 

823 else: 

824 keys = out["trial_id"].astype(str) 

825 lookup = metadata.frame.set_index(metadata.key_series()) 

826 for name in wanted: 

827 out[name] = keys.map(lookup[name]) 

828 return out 

829 

830 

831def trial_options_for(metadata: TrialMetadata | None, name: str) -> list[str]: 

832 """Sorted distinct values of a categorical trial field, for a multiselect.""" 

833 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

834 return [] 

835 return sorted({str(value) for value in metadata.joined_frame[name].dropna()}) 

836 

837 

838def trial_bounds_for( 

839 metadata: TrialMetadata | None, name: str 

840) -> tuple[float, float] | None: 

841 """``(min, max)`` of a numeric trial field over the loaded trials, or 

842 ``None`` when it has no range. 

843 

844 A field with one value over the loaded trials — a one-row table, or a pilot 

845 whose trials all share it — has no range, the rule :func:`bounds_for` and 

846 :func:`text_bounds_for` already follow. Returning ``(20.0, 20.0)`` handed 

847 the filter panel a slider Streamlit refuses to draw, which stopped the 

848 whole Scanpath view. 

849 """ 

850 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

851 return None 

852 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna() 

853 if numeric.empty: 

854 return None 

855 low, high = float(numeric.min()), float(numeric.max()) 

856 if low == high: 

857 return None 

858 return low, high 

859 

860 

861def trial_to_payload(metadata: TrialMetadata | None) -> dict | None: 

862 """Serialize the trial table for save & restore (DATA-29).""" 

863 if metadata is None or metadata.frame.empty: 

864 return None 

865 return { 

866 "source_name": metadata.source_name, 

867 "trial_column": metadata.trial_column, 

868 "participant_column": metadata.participant_column, 

869 "rows": metadata.frame.to_dict("records"), 

870 } 

871 

872 

873def trial_from_payload(payload: dict | None) -> TrialMetadata | None: 

874 """Rebuild a trial table from :func:`trial_to_payload`'s output.""" 

875 if not isinstance(payload, dict) or not payload.get("rows"): 

876 return None 

877 frame = pd.DataFrame(payload["rows"]) 

878 trial_column = str(payload.get("trial_column") or "trial_id") 

879 participant_column = payload.get("participant_column") 

880 if "trial_id" in frame.columns: 

881 # The payload holds the *clean* frame, whose key columns are already 

882 # canonical — rebuild against those rather than the original names. 

883 return build_trial_metadata( 

884 frame, 

885 "trial_id", 

886 "participant_id" if participant_column else None, 

887 source_name=str(payload.get("source_name") or "trial metadata"), 

888 ) 

889 return build_trial_metadata( 

890 frame, 

891 trial_column, 

892 participant_column, 

893 source_name=str(payload.get("source_name") or "trial metadata"), 

894 ) 

895 

896 

897def active_texts() -> TextMetadata | None: 

898 """The text table attached to this session, or ``None``.""" 

899 try: 

900 import streamlit as st 

901 

902 return st.session_state.get(TEXT_SESSION_KEY) 

903 except Exception: # no script run context (API, CLI, plain import) 

904 return None 

905 

906 

907def text_keys(combos: pd.DataFrame | None) -> set: 

908 """The distinct ``text_id`` values the loaded data actually has.""" 

909 if combos is None or combos.empty or "text_id" not in combos.columns: 

910 return set() 

911 return {str(value) for value in combos["text_id"].dropna().unique()} 

912 

913 

914def infer_text_id_column(frame: pd.DataFrame) -> str | None: 

915 """First plausible text-id column, or ``None`` — the UI's initial guess.""" 

916 if frame is None or frame.empty: 

917 return None 

918 lookup = {str(column).lower(): str(column) for column in frame.columns} 

919 for candidate in TEXT_ID_CANDIDATES: 

920 hit = lookup.get(candidate.lower()) 

921 if hit is not None: 

922 return hit 

923 return None 

924 

925 

926def _text_column_label(text_column) -> str: 

927 """Display form of a (possibly composite) text-column mapping — joined 

928 with " + ", matching how the wizard spells a composite trial id back out.""" 

929 return " + ".join(trial_mapping_columns(text_column)) 

930 

931 

932def build_text_metadata( 

933 frame: pd.DataFrame, 

934 text_column: str | list[str], 

935 *, 

936 source_name: str = "text metadata", 

937 keys: Iterable | None = None, 

938) -> TextMetadata: 

939 """Validate a raw text table into a registry + clean frame — third grain. 

940 

941 ``text_column`` is a single column name, or **several** to build a unique 

942 text id on the fly (joined with ``_``, like the Trial ID mapping the 

943 uploaded data itself uses — see :func:`data.trial_id_series`), the same 

944 composite-key trick :func:`build_trial_metadata` uses. The join/report 

945 logic underneath is flat, though — one dimension (``text_id``), never 

946 :func:`build_trial_metadata`'s participant-pairing option, since a text is 

947 a stimulus and nothing reads it as belonging to one reader. 

948 

949 ``keys`` is the set of text ids actually present in the loaded data; 

950 passing it fills in the two "unmatched" halves of the report. Rows whose 

951 id repeats are only a problem when they *disagree* — the rule every grain 

952 here follows. 

953 """ 

954 text_cols = trial_mapping_columns(text_column) 

955 label = _text_column_label(text_column) 

956 empty = TextMetadata(pd.DataFrame(columns=["text_id"]), (), source_name, label) 

957 if ( 

958 frame is None 

959 or frame.empty 

960 or not text_cols 

961 or any(c not in frame.columns for c in text_cols) 

962 ): 

963 return empty 

964 

965 work = _rows_with_ids(frame, text_cols) 

966 # See the matching comment in `build_trial_metadata` — the same "one 

967 # blank cell spells the id two ways" hazard applies to a text id. 

968 work["text_id"] = trial_id_series(work, text_column) 

969 if keys is not None: 

970 keys = list(keys) 

971 work["text_id"] = _respell_ids(work["text_id"], {str(tid) for tid in keys}) 

972 reserved = {*text_cols, "text_id", *_BOOKKEEPING_COLUMNS} 

973 value_columns = [ 

974 str(column) for column in frame.columns if str(column) not in reserved 

975 ] 

976 

977 work, _, duplicated, conflicting_set, combined = _merge_duplicates( 

978 work, work["text_id"], value_columns 

979 ) 

980 

981 fields: list[MetadataField] = [] 

982 clean = pd.DataFrame({"text_id": work["text_id"].to_numpy()}) 

983 for column in value_columns: 

984 dtype = _classify(work[column]) 

985 values = _coerce(work[column], dtype) 

986 clean[column] = values.to_numpy() 

987 fields.append( 

988 MetadataField( 

989 name=column, 

990 label=field_label(column), 

991 grain=GRAIN_TEXT, 

992 dtype=dtype, 

993 source=source_name, 

994 n_unique=int(values.dropna().nunique()), 

995 n_missing=int(values.isna().sum()), 

996 ) 

997 ) 

998 

999 table_ids = set(clean["text_id"]) | conflicting_set 

1000 if keys is None: 

1001 report = JoinReport( 

1002 matched=tuple(sorted(clean["text_id"])), 

1003 duplicated=duplicated, 

1004 conflicting=tuple(sorted(conflicting_set)), 

1005 combined_rows=combined, 

1006 ) 

1007 else: 

1008 data_ids = {str(tid) for tid in keys} 

1009 report = JoinReport( 

1010 matched=tuple(sorted(set(clean["text_id"]) & data_ids)), 

1011 only_in_table=tuple(sorted(table_ids - data_ids)), 

1012 only_in_data=tuple(sorted(data_ids - table_ids)), 

1013 duplicated=duplicated, 

1014 conflicting=tuple(sorted(conflicting_set)), 

1015 combined_rows=combined, 

1016 ) 

1017 return TextMetadata(clean, tuple(fields), source_name, label, report) 

1018 

1019 

1020def texts_matching( 

1021 metadata: TextMetadata | None, 

1022 selections: dict[str, Sequence] | None = None, 

1023 ranges: dict[str, tuple[float, float]] | None = None, 

1024 *, 

1025 keep_unknown: Mapping[str, bool] | None = None, 

1026) -> set | None: 

1027 """Text ids satisfying every metadata constraint, or ``None`` for "any". 

1028 

1029 Flat-grain sibling of :func:`participants_matching` — an empty constraint 

1030 must not narrow the pool to the texts *listed in the table*, and a numeric 

1031 range keeps a text with no value (``data.filter_trials``' rule that a 

1032 range narrows rather than excludes the unmeasured) unless ``keep_unknown`` 

1033 maps that field to ``False``. 

1034 """ 

1035 if metadata is None or metadata.frame.empty: 

1036 return None 

1037 active = { 

1038 name: list(values) 

1039 for name, values in (selections or {}).items() 

1040 if values and name in metadata.frame.columns 

1041 } 

1042 active_ranges = { 

1043 name: bounds 

1044 for name, bounds in (ranges or {}).items() 

1045 if bounds and name in metadata.frame.columns 

1046 } 

1047 if not active and not active_ranges: 

1048 return None 

1049 

1050 frame = metadata.frame 

1051 mask = pd.Series(True, index=frame.index) 

1052 for name, values in active.items(): 

1053 allowed = {str(value) for value in values} 

1054 mask &= frame[name].astype(str).isin(allowed) 

1055 mask &= _range_mask(frame, active_ranges, keep_unknown) 

1056 matching = set(frame.loc[mask, "text_id"]) 

1057 if not active and _keeps_unlisted(active_ranges, keep_unknown): 

1058 # Range-only narrowing keeps the unmeasured, including a text with no 

1059 # row at all (`participants_matching`'s rule, one grain over). 

1060 matching |= set(metadata.report.only_in_data) 

1061 return matching 

1062 

1063 

1064def project_texts( 

1065 metadata: TextMetadata | None, 

1066 frame: pd.DataFrame, 

1067 columns: Iterable[str] | None = None, 

1068) -> pd.DataFrame: 

1069 """Left-join chosen text-metadata columns onto a **small** text-keyed frame. 

1070 

1071 For ``combos`` — never for words or fixations, the rule every grain here 

1072 follows. Columns already present on ``frame`` win, so a recorded column is 

1073 never shadowed by a metadata field of the same name. 

1074 """ 

1075 if ( 

1076 metadata is None 

1077 or metadata.frame.empty 

1078 or frame is None 

1079 or frame.empty 

1080 or "text_id" not in frame.columns 

1081 ): 

1082 return frame 

1083 wanted = [ 

1084 name 

1085 for name in (list(columns) if columns is not None else list(metadata.names)) 

1086 if name in metadata.frame.columns and name not in frame.columns 

1087 ] 

1088 if not wanted: 

1089 return frame 

1090 out = frame.copy() 

1091 keys = out["text_id"].astype(str) 

1092 for name in wanted: 

1093 out[name] = keys.map(metadata.series(name)) 

1094 return out 

1095 

1096 

1097def text_options_for(metadata: TextMetadata | None, name: str) -> list[str]: 

1098 """Sorted distinct values of a categorical text field, for a multiselect.""" 

1099 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

1100 return [] 

1101 return sorted({str(value) for value in metadata.joined_frame[name].dropna()}) 

1102 

1103 

1104def text_bounds_for( 

1105 metadata: TextMetadata | None, name: str 

1106) -> tuple[float, float] | None: 

1107 """``(min, max)`` of a numeric text field, or ``None`` when it has no range.""" 

1108 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

1109 return None 

1110 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna() 

1111 if numeric.empty: 

1112 return None 

1113 low, high = float(numeric.min()), float(numeric.max()) 

1114 if low == high: 

1115 return None 

1116 return low, high 

1117 

1118 

1119def text_to_payload(metadata: TextMetadata | None) -> dict | None: 

1120 """Serialize the text table for save & restore.""" 

1121 if metadata is None or metadata.frame.empty: 

1122 return None 

1123 return { 

1124 "source_name": metadata.source_name, 

1125 "text_column": metadata.text_column, 

1126 "rows": metadata.frame.to_dict("records"), 

1127 } 

1128 

1129 

1130def text_from_payload(payload: dict | None) -> TextMetadata | None: 

1131 """Rebuild a text table from :func:`text_to_payload`'s output.""" 

1132 if not isinstance(payload, dict) or not payload.get("rows"): 

1133 return None 

1134 frame = pd.DataFrame(payload["rows"]) 

1135 text_column = str(payload.get("text_column") or "text_id") 

1136 if "text_id" in frame.columns: 

1137 # The payload holds the *clean* frame, whose key column is already 

1138 # canonical — rebuild against that rather than the original name. 

1139 return build_text_metadata( 

1140 frame, 

1141 "text_id", 

1142 source_name=str(payload.get("source_name") or "text metadata"), 

1143 ) 

1144 return build_text_metadata( 

1145 frame, 

1146 text_column, 

1147 source_name=str(payload.get("source_name") or "text metadata"), 

1148 ) 

1149 

1150 

1151def active() -> ParticipantMetadata | None: 

1152 """The participant table attached to this session, or ``None``. 

1153 

1154 Lives here rather than in the UI layer so the pure consumers 

1155 (:func:`project`, the filter resolution in ``controls``) can reach it 

1156 without importing Streamlit page code. Returns ``None`` outside a script 

1157 run, which is what the headless API and CLI see. 

1158 """ 

1159 try: 

1160 import streamlit as st 

1161 

1162 return st.session_state.get(SESSION_KEY) 

1163 except Exception: # no script run context (API, CLI, plain import) 

1164 return None 

1165 

1166 

1167def participant_ids(*frames: pd.DataFrame | None) -> list[str]: 

1168 """Every distinct reader id across the given frames, as sorted strings.""" 

1169 found: set = set() 

1170 for frame in frames: 

1171 if frame is None or frame.empty or "participant_id" not in frame.columns: 

1172 continue 

1173 found |= {str(value) for value in frame["participant_id"].dropna().unique()} 

1174 return sorted(found) 

1175 

1176 

1177def infer_participant_id_column(frame: pd.DataFrame) -> str | None: 

1178 """Best guess at the reader-id column, or ``None`` when nothing fits.""" 

1179 if frame is None or frame.empty: 

1180 return None 

1181 lowered = {str(column).lower(): str(column) for column in frame.columns} 

1182 for candidate in PARTICIPANT_ID_CANDIDATES: 

1183 if candidate in frame.columns: 

1184 return candidate 

1185 hit = lowered.get(candidate.lower()) 

1186 if hit is not None: 

1187 return hit 

1188 return None 

1189 

1190 

1191def _classify(series: pd.Series) -> str: 

1192 """Dtype bucket driving which control a field gets (range vs membership).""" 

1193 cleaned = series.dropna() 

1194 if cleaned.empty: 

1195 return _DTYPE_CATEGORICAL 

1196 if pd.api.types.is_bool_dtype(cleaned): 

1197 return _DTYPE_BOOLEAN 

1198 if pd.api.types.is_numeric_dtype(cleaned): 

1199 return _DTYPE_NUMERIC 

1200 # A column of numeric strings ("23", "4.5") is numeric in every way the user 

1201 # cares about; anything else stays categorical rather than being coerced. 

1202 numeric = pd.to_numeric(cleaned, errors="coerce") 

1203 if numeric.notna().all(): 

1204 return _DTYPE_NUMERIC 

1205 return _DTYPE_CATEGORICAL 

1206 

1207 

1208def _coerce(series: pd.Series, dtype: str) -> pd.Series: 

1209 if dtype == _DTYPE_NUMERIC: 

1210 # An infinite value is no value (round 10): the record counts as 

1211 # unknown, which a range keeps unless *Keep unknown values* is off — 

1212 # and a slider never gets an infinite end. All three grains read here. 

1213 numbers = pd.to_numeric(series, errors="coerce") 

1214 infinite = np.isinf(numbers.astype(float)) 

1215 return numbers.mask(infinite) if infinite.any() else numbers 

1216 if dtype == _DTYPE_BOOLEAN: 

1217 return series 

1218 return series.astype("object").where(series.notna(), np.nan) 

1219 

1220 

1221def field_label(name: str) -> str: 

1222 """The label for a metadata field: its column name, as the table spelled it. 

1223 

1224 DATA-66: a field the user attached is shown under the name it has in their 

1225 file (``native_language``, not "Native language"). Public because it is the 

1226 *only* labeller for a metadata field: the picker in ``tabs._pretty_col`` has 

1227 to name a field the same way whether or not it can reach the attached table 

1228 at that moment. 

1229 """ 

1230 return str(name) 

1231 

1232 

1233def build_participant_metadata( 

1234 frame: pd.DataFrame, 

1235 id_column: str, 

1236 *, 

1237 source_name: str = "participant metadata", 

1238 participants: Iterable | None = None, 

1239) -> ParticipantMetadata: 

1240 """Validate a raw participant table into a registry + clean frame. 

1241 

1242 ``participants`` is the set of reader ids actually present in the loaded 

1243 data; passing it fills in the two "unmatched" halves of the report. Rows 

1244 whose id repeats are only a problem when they *disagree* — duplicated rows 

1245 that do not are combined field by field (:func:`_merge_duplicates`, counted 

1246 in ``report.combined_rows``), duplicated rows that say something different 

1247 are dropped and reported. 

1248 """ 

1249 if frame is None or frame.empty or id_column not in frame.columns: 

1250 return ParticipantMetadata( 

1251 pd.DataFrame(columns=["participant_id"]), (), source_name, str(id_column) 

1252 ) 

1253 

1254 work = _rows_with_ids(frame, [id_column]) 

1255 # See the matching comment in `build_trial_metadata` — the same "one blank 

1256 # cell spells the id two ways" hazard applies to a reader id. 

1257 work["participant_id"] = stable_id(work[id_column]) 

1258 if participants is not None: 

1259 work["participant_id"] = _match_padding(work["participant_id"], participants) 

1260 value_columns = [ 

1261 str(column) 

1262 for column in frame.columns 

1263 if str(column) not in {str(id_column), "participant_id", *_BOOKKEEPING_COLUMNS} 

1264 ] 

1265 

1266 work, _, duplicated, conflicting_set, combined = _merge_duplicates( 

1267 work, work["participant_id"], value_columns 

1268 ) 

1269 

1270 fields: list[MetadataField] = [] 

1271 clean = pd.DataFrame({"participant_id": work["participant_id"].to_numpy()}) 

1272 for column in value_columns: 

1273 dtype = _classify(work[column]) 

1274 values = _coerce(work[column], dtype) 

1275 clean[column] = values.to_numpy() 

1276 fields.append( 

1277 MetadataField( 

1278 name=column, 

1279 label=field_label(column), 

1280 grain=GRAIN_PARTICIPANT, 

1281 dtype=dtype, 

1282 source=source_name, 

1283 n_unique=int(values.dropna().nunique()), 

1284 n_missing=int(values.isna().sum()), 

1285 ) 

1286 ) 

1287 

1288 table_ids = set(clean["participant_id"]) | conflicting_set 

1289 if participants is None: 

1290 report = JoinReport( 

1291 matched=tuple(sorted(clean["participant_id"])), 

1292 duplicated=duplicated, 

1293 conflicting=tuple(sorted(conflicting_set)), 

1294 combined_rows=combined, 

1295 ) 

1296 else: 

1297 data_ids = {str(pid) for pid in participants} 

1298 report = JoinReport( 

1299 matched=tuple(sorted(set(clean["participant_id"]) & data_ids)), 

1300 only_in_table=tuple(sorted(table_ids - data_ids)), 

1301 only_in_data=tuple(sorted(data_ids - table_ids)), 

1302 duplicated=duplicated, 

1303 conflicting=tuple(sorted(conflicting_set)), 

1304 combined_rows=combined, 

1305 ) 

1306 return ParticipantMetadata( 

1307 clean, tuple(fields), source_name, str(id_column), report 

1308 ) 

1309 

1310 

1311def _match_padding(ids: pd.Series, participants: Iterable) -> pd.Series: 

1312 """``ids`` spelled the data's way when only zero-padding differs (BUG-59). 

1313 

1314 A metadata CSV reads a reader ``007`` as the number 7 while the data kept 

1315 "007", and the table then joined to no one; ``data.zero_padding_map`` 

1316 decides, and refuses whenever the match is not unambiguous. 

1317 """ 

1318 return _respell_ids(ids, {str(pid) for pid in participants}) 

1319 

1320 

1321def _respell_ids(ids: pd.Series, reference: set) -> pd.Series: 

1322 """``ids`` spelled the way ``reference`` spells them, when the two differ 

1323 only by zero-padding (BUG-59) or by the escaping composite ids gained 

1324 (``data.composite_respelling_map``: a dataset restored from the recovery 

1325 cache keeps the ids it was stored with, a table attached today composes 

1326 them anew). Both maps refuse anything ambiguous.""" 

1327 unique = ids.unique() 

1328 mapping = zero_padding_map(unique, reference) or composite_respelling_map( 

1329 unique, reference 

1330 ) 

1331 return ids.replace(mapping) if mapping else ids 

1332 

1333 

1334def rejoin( 

1335 metadata: ParticipantMetadata, participants: Iterable 

1336) -> ParticipantMetadata: 

1337 """Recompute the join report against a (possibly new) participant list.""" 

1338 data_ids = {str(pid) for pid in participants} 

1339 if not metadata.frame.empty: 

1340 renamed = _match_padding(metadata.frame["participant_id"], data_ids) 

1341 if not renamed.equals(metadata.frame["participant_id"]): 

1342 metadata = replace( 

1343 metadata, frame=metadata.frame.assign(participant_id=renamed) 

1344 ) 

1345 usable_ids = ( 

1346 set(metadata.frame["participant_id"]) if not metadata.frame.empty else set() 

1347 ) 

1348 # Conflicting ids are *in the table* — so they are not "only in the data" — 

1349 # but they carry no values, so they are not joined either. Counting them as 

1350 # matched (as an earlier version did) made "Joined to N readers" grow by the 

1351 # conflict count on the first rerun after the file was attached, disagreeing 

1352 # with what `build_participant_metadata` had just reported. 

1353 table_ids = usable_ids | set(metadata.report.conflicting) 

1354 return ParticipantMetadata( 

1355 metadata.frame, 

1356 metadata.fields, 

1357 metadata.source_name, 

1358 metadata.id_column, 

1359 JoinReport( 

1360 matched=tuple(sorted(usable_ids & data_ids)), 

1361 only_in_table=tuple(sorted(table_ids - data_ids)), 

1362 only_in_data=tuple(sorted(data_ids - table_ids)), 

1363 duplicated=metadata.report.duplicated, 

1364 conflicting=metadata.report.conflicting, 

1365 combined_rows=metadata.report.combined_rows, 

1366 ), 

1367 ) 

1368 

1369 

1370def participants_matching( 

1371 metadata: ParticipantMetadata | None, 

1372 selections: dict[str, Sequence] | None = None, 

1373 ranges: dict[str, tuple[float, float]] | None = None, 

1374 *, 

1375 keep_unknown: Mapping[str, bool] | None = None, 

1376) -> set | None: 

1377 """Reader ids satisfying every metadata constraint, or ``None`` for "any". 

1378 

1379 Returning ``None`` rather than "all ids" is deliberate: an empty constraint 

1380 must not narrow the pool to the readers *listed in the table*, which would 

1381 quietly drop everyone the table forgot. 

1382 

1383 Membership follows the categorical filters; a numeric range keeps readers 

1384 with **no value**, matching ``data.filter_trials``' rule that a range is a 

1385 narrowing control and not an exclusion of the unmeasured. ``keep_unknown`` 

1386 maps a ranged field to ``False`` to leave those readers out instead — the 

1387 explicit *Keep unknown values* choice beside the slider. 

1388 """ 

1389 if metadata is None or metadata.frame.empty: 

1390 return None 

1391 active = { 

1392 name: list(values) 

1393 for name, values in (selections or {}).items() 

1394 if values and name in metadata.frame.columns 

1395 } 

1396 active_ranges = { 

1397 name: bounds 

1398 for name, bounds in (ranges or {}).items() 

1399 if bounds and name in metadata.frame.columns 

1400 } 

1401 if not active and not active_ranges: 

1402 return None 

1403 

1404 frame = metadata.frame 

1405 mask = pd.Series(True, index=frame.index) 

1406 for name, values in active.items(): 

1407 allowed = {str(value) for value in values} 

1408 mask &= frame[name].astype(str).isin(allowed) 

1409 mask &= _range_mask(frame, active_ranges, keep_unknown) 

1410 matching = set(frame.loc[mask, "participant_id"]) 

1411 if not active and _keeps_unlisted(active_ranges, keep_unknown): 

1412 # Range-only narrowing keeps the unmeasured (`data.filter_trials`' rule, 

1413 # UX-49) — and a reader with **no row at all** is the most unmeasured 

1414 # there is, so they are kept on the same terms as a reader whose value 

1415 # is NaN. A *categorical* selection still excludes them, matching every 

1416 # other membership filter in the app: "only Hebrew speakers" cannot 

1417 # include a reader whose language is unknown. 

1418 matching |= set(metadata.report.only_in_data) 

1419 return matching 

1420 

1421 

1422def project( 

1423 metadata: ParticipantMetadata | None, 

1424 frame: pd.DataFrame, 

1425 columns: Iterable[str] | None = None, 

1426) -> pd.DataFrame: 

1427 """Left-join chosen metadata columns onto a **small** participant-keyed frame. 

1428 

1429 For ``combos`` (one row per trial) and group-by results — never for words or 

1430 fixations. Columns already present on ``frame`` win, so a real recorded 

1431 column is never shadowed by a metadata field of the same name. 

1432 """ 

1433 if ( 

1434 metadata is None 

1435 or metadata.frame.empty 

1436 or frame is None 

1437 or frame.empty 

1438 or "participant_id" not in frame.columns 

1439 ): 

1440 return frame 

1441 wanted = [ 

1442 name 

1443 for name in (list(columns) if columns is not None else list(metadata.names)) 

1444 if name in metadata.frame.columns and name not in frame.columns 

1445 ] 

1446 if not wanted: 

1447 return frame 

1448 out = frame.copy() 

1449 keys = out["participant_id"].astype(str) 

1450 for name in wanted: 

1451 out[name] = keys.map(metadata.series(name)) 

1452 return out 

1453 

1454 

1455def options_for(metadata: ParticipantMetadata | None, name: str) -> list[str]: 

1456 """Sorted distinct values of a categorical field, for a multiselect. 

1457 

1458 Built from :attr:`ParticipantMetadata.joined_frame` — the loaded readers 

1459 only — so the control cannot offer a value that matches nobody. 

1460 """ 

1461 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

1462 return [] 

1463 values = metadata.joined_frame[name].dropna() 

1464 return sorted({str(value) for value in values}) 

1465 

1466 

1467def bounds_for( 

1468 metadata: ParticipantMetadata | None, name: str 

1469) -> tuple[float, float] | None: 

1470 """``(min, max)`` of a numeric field, or ``None`` when it has no range.""" 

1471 if metadata is None or metadata.frame.empty or name not in metadata.frame.columns: 

1472 return None 

1473 numeric = pd.to_numeric(metadata.joined_frame[name], errors="coerce").dropna() 

1474 if numeric.empty: 

1475 return None 

1476 low, high = float(numeric.min()), float(numeric.max()) 

1477 if low == high: 

1478 return None 

1479 return low, high 

1480 

1481 

1482# ----------------------------------------------------------------------------- 

1483# Serialization — the ENG-26 on-device recovery cache (DATA-38, see 

1484# `session_payloads` below) and the payload `api`/`cli` hand in. (The 💾 Session 

1485# backup carried these too until UX-179 cut the settings file to the figure.) Records rather than a pickled frame, so it round-trips 

1486# through JSON like every other saved setting. 

1487# ----------------------------------------------------------------------------- 

1488 

1489 

1490def to_payload(metadata: ParticipantMetadata | None) -> dict | None: 

1491 if metadata is None or metadata.frame.empty: 

1492 return None 

1493 return { 

1494 "grain": GRAIN_PARTICIPANT, 

1495 "id_column": metadata.id_column, 

1496 "source_name": metadata.source_name, 

1497 "records": metadata.frame.to_dict(orient="records"), 

1498 } 

1499 

1500 

1501def from_payload(payload: dict | None) -> ParticipantMetadata | None: 

1502 if not payload or not payload.get("records"): 

1503 return None 

1504 frame = pd.DataFrame(payload["records"]) 

1505 if "participant_id" not in frame.columns: 

1506 return None 

1507 return build_participant_metadata( 

1508 frame, 

1509 "participant_id", 

1510 source_name=str(payload.get("source_name") or "participant metadata"), 

1511 ) 

1512 

1513 

1514# ----------------------------------------------------------------------------- 

1515# DATA-38 — attached tables in the ENG-26 on-device recovery cache, and DATA-47 — 

1516# the tables belong to a dataset. 

1517# 

1518# The session keys above hold the tables of the *selected* dataset only — the 

1519# one every consumer (filters, chips, sort, inspection, export) reads through 

1520# `active()` / `active_trials()` / `active_texts()`. Every other dataset's 

1521# tables wait in a per-dataset store, and `activate_dataset` swaps them in and 

1522# out when the selection changes. They used to be one slot per grain for the 

1523# whole session, so a new dataset opened with the last one's tables, attaching a 

1524# table to dataset B replaced dataset A's, and detaching it anywhere removed it 

1525# everywhere. The cache writes the store, keyed by dataset. 

1526# ----------------------------------------------------------------------------- 

1527 

1528#: What a grain's ``*_FILE_SESSION_KEY`` holds when its table came back from the 

1529#: recovery cache or a saved config, rather than from a file in the uploader. 

1530#: The metadata sections read an empty uploader as "the user just removed the 

1531#: file" and detach on sight (UX-115) — and a restored table has no file in the 

1532#: uploader, so without this marker the first visit to the 🗂️ Data page would 

1533#: detach exactly what the restore brought back. 

1534RESTORED_FILE_SIGNATURE = "restored" 

1535 

1536#: ``(grain, table key, raw key, file key, to_payload, from_payload)`` per grain. 

1537_GRAINS = ( 

1538 ( 

1539 GRAIN_PARTICIPANT, 

1540 SESSION_KEY, 

1541 RAW_SESSION_KEY, 

1542 FILE_SESSION_KEY, 

1543 to_payload, 

1544 from_payload, 

1545 ), 

1546 ( 

1547 "trial", 

1548 TRIAL_SESSION_KEY, 

1549 TRIAL_RAW_SESSION_KEY, 

1550 TRIAL_FILE_SESSION_KEY, 

1551 trial_to_payload, 

1552 trial_from_payload, 

1553 ), 

1554 ( 

1555 "text", 

1556 TEXT_SESSION_KEY, 

1557 TEXT_RAW_SESSION_KEY, 

1558 TEXT_FILE_SESSION_KEY, 

1559 text_to_payload, 

1560 text_from_payload, 

1561 ), 

1562) 

1563_GRAIN_KEYS = {grain: (key, raw, file) for grain, key, raw, file, *_ in _GRAINS} 

1564 

1565 

1566#: The payloads' row lists — `to_payload` says ``records``, the other two ``rows``. 

1567_ROW_KEYS = ("records", "rows") 

1568 

1569 

1570def session_payloads(session) -> dict[str, dict]: 

1571 """Every attached table as its save & restore payload, keyed by grain. 

1572 

1573 Each carries its frame's ``columns`` too: the cache writes its manifest with 

1574 sorted keys, which would otherwise hand the rows back alphabetised and 

1575 reorder the table's fields everywhere they are listed. 

1576 """ 

1577 payloads = {} 

1578 for grain, key, _raw, _file, dump, _load in _GRAINS: 

1579 attached = session.get(key) 

1580 payload = dump(attached) 

1581 if payload is not None: 

1582 payloads[grain] = {**payload, "columns": list(attached.frame.columns)} 

1583 return payloads 

1584 

1585 

1586def _in_column_order(payload): 

1587 """``payload`` with each row's keys back in its ``columns`` order.""" 

1588 columns = payload.get("columns") if isinstance(payload, dict) else None 

1589 if not columns: 

1590 return payload 

1591 ordered = dict(payload) 

1592 for rows_key in _ROW_KEYS: 

1593 rows = payload.get(rows_key) 

1594 if isinstance(rows, list): 

1595 ordered[rows_key] = [ 

1596 {column: row[column] for column in columns if column in row} 

1597 for row in rows 

1598 if isinstance(row, dict) 

1599 ] 

1600 return ordered 

1601 

1602 

1603def session_signature(session) -> list: 

1604 """A cheap content fingerprint of the attached tables. 

1605 

1606 For the recovery cache's every-rerun "did anything change" check. Object 

1607 identity will not do: the tables are rebuilt on every render of the Data 

1608 page, and the participant one is re-joined on every run, so a new object 

1609 arrives when nothing changed. The frames are small (one row per reader, 

1610 trial or text), so hashing their content is cheap. 

1611 """ 

1612 signature = [] 

1613 for grain, key, *_ in _GRAINS: 

1614 attached = session.get(key) 

1615 frame = getattr(attached, "frame", None) 

1616 if not isinstance(frame, pd.DataFrame) or frame.empty: 

1617 continue 

1618 try: 

1619 # Row hashes in row order — a sum would miss a reordered table. 

1620 cells = pd.util.hash_pandas_object(frame, index=False).to_numpy().tobytes() 

1621 except (TypeError, ValueError): # unhashable cells — hash their text 

1622 cells = frame.to_csv(index=False).encode("utf-8") 

1623 digest = hashlib.sha256(cells).hexdigest() 

1624 signature.append( 

1625 [ 

1626 grain, 

1627 str(getattr(attached, "source_name", "")), 

1628 list(frame.columns), 

1629 digest, 

1630 ] 

1631 ) 

1632 return signature 

1633 

1634 

1635def grain_keys(grain: str) -> tuple[str, str, str]: 

1636 """``(table key, raw key, file key)`` in session state for ``grain``.""" 

1637 return _GRAIN_KEYS[grain] 

1638 

1639 

1640def mark_restored(session, grain: str, attached) -> None: 

1641 """Attach ``attached`` as a table with no live upload behind it. 

1642 

1643 The recovery cache hands back a table the uploader never saw — see 

1644 :data:`RESTORED_FILE_SIGNATURE`. 

1645 """ 

1646 key, raw, file = _GRAIN_KEYS[grain] 

1647 session[key] = attached 

1648 session[raw] = attached.frame 

1649 session[file] = RESTORED_FILE_SIGNATURE 

1650 

1651 

1652def is_restored(session, grain: str) -> bool: 

1653 """Whether ``grain``'s attached table came back without a file behind it.""" 

1654 return session.get(_GRAIN_KEYS[grain][2]) == RESTORED_FILE_SIGNATURE 

1655 

1656 

1657def restore_payloads(session, payloads) -> int: 

1658 """Re-attach the tables :func:`session_payloads` wrote; how many landed. 

1659 

1660 A grain already attached in this session keeps its own table — the same 

1661 "never overwrite what is already seeded" rule the rest of the restore 

1662 follows — and a payload that no longer builds is skipped, not raised: a 

1663 stale cache must never stop the app opening. 

1664 """ 

1665 if not isinstance(payloads, dict): 

1666 return 0 

1667 restored = 0 

1668 for grain, key, _raw, _file, _dump, load in _GRAINS: 

1669 if session.get(key) is not None: 

1670 continue 

1671 try: 

1672 attached = load(_in_column_order(payloads.get(grain))) 

1673 except (ValueError, TypeError, KeyError): 

1674 attached = None 

1675 if attached is None: 

1676 continue 

1677 mark_restored(session, grain, attached) 

1678 restored += 1 

1679 return restored 

1680 

1681 

1682#: DATA-47 — every dataset's tables but the selected one's, as the payloads 

1683#: :func:`session_payloads` builds: ``{dataset: {grain: payload}}``. Payloads 

1684#: rather than table objects so the cache can write them as they are. 

1685DATASET_STORE_KEY = "_metadata_by_dataset" 

1686#: Which dataset the session keys' tables belong to right now. 

1687OWNER_KEY = "_metadata_owner" 

1688#: Bumped on every change to the store — the cache's cheap "did it change" test, 

1689#: since hashing every stored table on every rerun would not be cheap. 

1690STORE_REVISION_KEY = "_metadata_store_revision" 

1691#: The add-dataset wizard's dataset, before it has a name. Never cached. 

1692PENDING_DATASET = "\x00pending" 

1693#: Bumped by :func:`reset_uploads`; part of every metadata uploader's key. 

1694UPLOAD_GENERATION_KEY = "_metadata_upload_generation" 

1695 

1696 

1697def upload_key(grain: str, session=None) -> str: 

1698 """``grain``'s uploader widget key, in the current upload generation. 

1699 

1700 Popping an uploader's key from session state does not empty it: the browser 

1701 still holds the file and sends it back on the next rerun, where it reads as 

1702 a new upload — so a table attached to one dataset re-attached itself to the 

1703 next one opened. Only a new key gives the browser a fresh, empty uploader, 

1704 so a dataset switch moves every uploader to a new generation. 

1705 """ 

1706 if session is None: 

1707 import streamlit as st 

1708 

1709 session = st.session_state 

1710 generation = int(session.get(UPLOAD_GENERATION_KEY) or 0) 

1711 base = f"{grain}_metadata_upload" 

1712 return base if generation == 0 else f"{base}_{generation}" 

1713 

1714 

1715def reset_uploads(session) -> None: 

1716 """Empty every metadata uploader, in the browser too (:func:`upload_key`).""" 

1717 for grain, *_ in _GRAINS: 

1718 session.pop(upload_key(grain, session), None) 

1719 session[UPLOAD_GENERATION_KEY] = int(session.get(UPLOAD_GENERATION_KEY) or 0) + 1 

1720 

1721 

1722def _widget_keys(grain: str) -> tuple[str, ...]: 

1723 """The UI state of ``grain``'s section that describes one dataset's table. 

1724 

1725 The display name, the id-column and keep-fields picks, so the next 

1726 dataset's table starts from its own auto-detect. The uploader is emptied 

1727 separately, by :func:`reset_uploads`. 

1728 """ 

1729 return ( 

1730 f"_{grain}_metadata_name", 

1731 f"{grain}_metadata_id_column", 

1732 f"{grain}_metadata_keep_fields", 

1733 ) 

1734 

1735 

1736def clear_active(session) -> None: 

1737 """Detach the selected dataset's tables from the session keys — all grains.""" 

1738 for grain, key, raw, file, *_ in _GRAINS: 

1739 for name in (key, raw, file, *_widget_keys(grain)): 

1740 session.pop(name, None) 

1741 reset_uploads(session) 

1742 

1743 

1744def _store(session) -> dict: 

1745 store = session.get(DATASET_STORE_KEY) 

1746 return dict(store) if isinstance(store, dict) else {} 

1747 

1748 

1749def _set_store(session, store: dict) -> None: 

1750 session[DATASET_STORE_KEY] = store 

1751 session[STORE_REVISION_KEY] = int(session.get(STORE_REVISION_KEY) or 0) + 1 

1752 

1753 

1754def stash_active(session) -> None: 

1755 """File the session keys' tables under the dataset they belong to.""" 

1756 owner = session.get(OWNER_KEY) 

1757 if owner is None: 

1758 return 

1759 store = _store(session) 

1760 payloads = session_payloads(session) 

1761 if payloads: 

1762 if store.get(owner) == payloads: 

1763 return # unchanged since it was restored — nothing for the cache to do 

1764 store[owner] = payloads 

1765 elif owner not in store: 

1766 return 

1767 else: 

1768 store.pop(owner) 

1769 _set_store(session, store) 

1770 

1771 

1772def activate_dataset(session, dataset: str) -> bool: 

1773 """Make ``dataset``'s tables the attached ones; whether anything moved. 

1774 

1775 Called by ``app.main`` on every run with the selected dataset. When the 

1776 selection changed, the outgoing dataset's tables are filed away, the session 

1777 keys are cleared — widgets included, see :func:`_widget_keys` — and the 

1778 incoming dataset's are restored (marked restored, since no uploader holds 

1779 their file). A session whose tables have no owner yet (its first run) adopts 

1780 whatever is attached for ``dataset`` rather than clearing it. 

1781 """ 

1782 dataset = str(dataset) 

1783 owner = session.get(OWNER_KEY) 

1784 if owner == dataset: 

1785 return False 

1786 if owner is not None: 

1787 stash_active(session) 

1788 clear_active(session) 

1789 session[OWNER_KEY] = dataset 

1790 restore_payloads(session, _store(session).get(dataset)) 

1791 return True 

1792 

1793 

1794#: CMP-8's key prefix for scanpath B's filters, and the picker's "same dataset" 

1795#: answer (``compare_source.THIS_DATASET`` — not imported: `compare_source` 

1796#: imports `app`, which imports this). 

1797_COMPARE_PREFIX = "cmp" 

1798_COMPARE_SAME_DATASET = "This dataset" 

1799_BUILT_KEY = "_metadata_built_for_compare" 

1800 

1801 

1802#: EXP-22 — the table each grain is named by in a title / caption pattern: 

1803#: ``{trials.font_size}`` is the trial table's ``font_size``. 

1804PATTERN_TABLE_NAMES = { 

1805 GRAIN_PARTICIPANT: "participants", 

1806 GRAIN_TRIAL: "trials", 

1807 GRAIN_TEXT: "texts", 

1808} 

1809 

1810 

1811def pattern_rows( 

1812 participant, trial, text_id=None, *, prefix: str = "" 

1813) -> dict[str, dict[str, object]]: 

1814 """This trial's row of every attached metadata table (EXP-22). 

1815 

1816 ``{"participants": {...}, "trials": {...}, "texts": {...}}`` — only the 

1817 tables that are attached, each with every registered field (a reader, trial 

1818 or text the table does not mention gets ``None`` values, so the field still 

1819 exists and a pattern naming it still validates). What 

1820 ``export.table_pattern_fields`` turns into ``{trials.font_size}``. 

1821 """ 

1822 rows: dict[str, dict[str, object]] = {} 

1823 table = attached_for(GRAIN_PARTICIPANT, prefix) 

1824 if table is not None: 

1825 found = table.values_for(participant) if participant is not None else {} 

1826 rows["participants"] = {name: found.get(name) for name in table.names} 

1827 table = attached_for(GRAIN_TRIAL, prefix) 

1828 if table is not None: 

1829 found: dict = {} 

1830 if trial is not None and not table.frame.empty: 

1831 match = table.frame["trial_id"] == str(trial) 

1832 if table.keyed_by_participant: 

1833 match &= table.frame["participant_id"] == str(participant) 

1834 hit = table.frame[match] 

1835 if not hit.empty: 

1836 found = hit.iloc[0].to_dict() 

1837 rows["trials"] = {name: found.get(name) for name in table.names} 

1838 table = attached_for(GRAIN_TEXT, prefix) 

1839 if table is not None: 

1840 found = table.values_for(text_id) if text_id is not None else {} 

1841 rows["texts"] = {name: found.get(name) for name in table.names} 

1842 return rows 

1843 

1844 

1845def attached_for(grain: str, prefix: str = ""): 

1846 """The table ``grain``'s filters under key ``prefix`` narrow by (DATA-47). 

1847 

1848 The main pool's filters read the selected dataset's table. Compare mode's 

1849 scanpath B (the ``cmp`` prefix) can come from another dataset, and then its 

1850 filters must read *that* dataset's own table — which waits in the store — 

1851 not A's. Built from the stored payload once per store revision. 

1852 """ 

1853 try: 

1854 import streamlit as st 

1855 

1856 session = st.session_state 

1857 live = session.get(_GRAIN_KEYS[grain][0]) 

1858 except Exception: # no script run context (API, CLI, plain import) 

1859 return None 

1860 if prefix != _COMPARE_PREFIX: 

1861 return live 

1862 other = session.get(COMPARE_SOURCE_STATE_KEY) 

1863 if ( 

1864 not other 

1865 or other == _COMPARE_SAME_DATASET 

1866 or str(other) == session.get(OWNER_KEY) 

1867 ): 

1868 return live 

1869 payload = (_store(session).get(str(other)) or {}).get(grain) 

1870 if not isinstance(payload, dict): 

1871 return None 

1872 revision = session.get(STORE_REVISION_KEY) 

1873 built = session.get(_BUILT_KEY) 

1874 cache_key = (str(other), grain, revision) 

1875 if not isinstance(built, dict) or cache_key not in built: 

1876 load = next(entry[-1] for entry in _GRAINS if entry[0] == grain) 

1877 try: 

1878 table = load(_in_column_order(payload)) 

1879 except (ValueError, TypeError, KeyError): 

1880 table = None 

1881 kept = { 

1882 k: v 

1883 for k, v in (built or {}).items() 

1884 if isinstance(k, tuple) and k[-1] == revision 

1885 } 

1886 session[_BUILT_KEY] = built = {**kept, cache_key: table} 

1887 return built[cache_key] 

1888 

1889 

1890def begin_pending_dataset(session) -> None: 

1891 """Start the add-dataset wizard's dataset with no tables of its own.""" 

1892 store = _store(session) 

1893 if PENDING_DATASET in store: 

1894 store.pop(PENDING_DATASET) 

1895 _set_store(session, store) 

1896 

1897 

1898def adopt_pending_dataset(session, dataset: str) -> None: 

1899 """✅ Add dataset: the wizard's tables become ``dataset``'s. 

1900 

1901 The session keys already hold them (the deferred join has just attached 

1902 them), so this only renames who owns them — the next run's 

1903 :func:`activate_dataset` then sees nothing to swap. 

1904 """ 

1905 begin_pending_dataset(session) 

1906 session[OWNER_KEY] = str(dataset) 

1907 

1908 

1909def forget_dataset(session, dataset: str) -> None: 

1910 """A removed dataset's tables go with it.""" 

1911 dataset = str(dataset) 

1912 store = _store(session) 

1913 if dataset in store: 

1914 store.pop(dataset) 

1915 _set_store(session, store) 

1916 if session.get(OWNER_KEY) == dataset: 

1917 clear_active(session) 

1918 session.pop(OWNER_KEY, None) 

1919 

1920 

1921def rename_dataset(session, old: str, new: str) -> None: 

1922 """A renamed dataset keeps its tables.""" 

1923 old, new = str(old), str(new) 

1924 store = _store(session) 

1925 if old in store: 

1926 _set_store(session, {(new if k == old else k): v for k, v in store.items()}) 

1927 if session.get(OWNER_KEY) == old: 

1928 session[OWNER_KEY] = new 

1929 

1930 

1931def dataset_payloads(session) -> dict[str, dict]: 

1932 """Every dataset's tables, the selected one's live: ``{dataset: {grain: …}}``. 

1933 

1934 What the recovery cache writes. The add-dataset wizard's unnamed dataset is 

1935 left out — it is not a dataset yet, and a restart discards the wizard. 

1936 """ 

1937 store = _store(session) 

1938 owner = session.get(OWNER_KEY) 

1939 if owner is not None: 

1940 live = session_payloads(session) 

1941 if live: 

1942 store[owner] = live 

1943 else: 

1944 store.pop(owner, None) 

1945 store.pop(PENDING_DATASET, None) 

1946 return {name: payloads for name, payloads in store.items() if payloads} 

1947 

1948 

1949def store_signature(session) -> list: 

1950 """A cheap fingerprint of every dataset's tables, for the cache (DATA-47). 

1951 

1952 The live tables by content (:func:`session_signature` — they are rebuilt on 

1953 every render), the rest by the store's revision counter. Empty when nothing 

1954 is attached anywhere, which is what tells the cache to delete its file. 

1955 """ 

1956 live = session_signature(session) 

1957 # Deliberately not `dataset_payloads`, which serializes the live tables: this 

1958 # runs on every rerun, and the live half is already covered by `live`. 

1959 owner = session.get(OWNER_KEY) 

1960 stored = any( 

1961 tables 

1962 for name, tables in _store(session).items() 

1963 if name not in (owner, PENDING_DATASET) 

1964 ) 

1965 if not live and not stored: 

1966 return [] 

1967 return [ 

1968 ["store", int(session.get(STORE_REVISION_KEY) or 0)], 

1969 ["owner", str(session.get(OWNER_KEY))], 

1970 *live, 

1971 ] 

1972 

1973 

1974def restore_dataset_payloads(session, payloads) -> int: 

1975 """Put the tables :func:`dataset_payloads` wrote back in the store; how many. 

1976 

1977 A dataset this session already holds tables for keeps its own. The selected 

1978 dataset's go straight onto the session keys, grain by grain — one already 

1979 attached is kept, like the rest of the restore's ``setdefault`` — and the 

1980 others wait in the store for :func:`activate_dataset`. 

1981 """ 

1982 datasets = payloads.get("datasets") if isinstance(payloads, dict) else None 

1983 if not isinstance(datasets, dict): 

1984 return 0 

1985 store = _store(session) 

1986 owner = session.get(OWNER_KEY) 

1987 restored = 0 

1988 changed = False 

1989 for name, tables in datasets.items(): 

1990 name = str(name) 

1991 if not isinstance(tables, dict) or name == PENDING_DATASET or name in store: 

1992 continue 

1993 store[name] = tables 

1994 changed = True 

1995 if name == owner: 

1996 restored += restore_payloads(session, tables) 

1997 else: 

1998 restored += sum( 

1999 1 for grain, *_ in _GRAINS if isinstance(tables.get(grain), dict) 

2000 ) 

2001 if changed: 

2002 _set_store(session, store) 

2003 return restored