Coverage for scanpath_studio/column_names.py: 97%

436 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""DATA-66: the dataset's own column names, behind the canonical ones. 

2 

3Normalization (`data.normalize_*`) rebuilds every table under canonical names — 

4`duration_ms`, `total_fixation_duration_ms` — and, until this module, kept no 

5record of what each was called in the user's files. A :class:`ColumnNames` is 

6that record, one per table of a dataset: for each canonical column, the source 

7column(s) it was read from and how (`kind`). :func:`from_schema` builds it from 

8the same schema, registries and raw columns normalization used, so the two 

9cannot disagree. Pure — no Streamlit, no I/O. 

10 

11The canonical names stay the internal contract and every wire format's values; 

12this map is what a person is shown, what an export writes and what the API 

13accepts (DATA-66 phases 2–4). 

14""" 

15 

16from __future__ import annotations 

17 

18import json 

19from collections.abc import Iterable, Mapping 

20from dataclasses import dataclass, field 

21 

22from .constants import SAMPLE_INDEX 

23 

24#: How a canonical column came to be. 

25MAPPED = "mapped" # read from one source column, unchanged 

26COMPOSITE = "composite" # several source columns joined (a composite id) 

27CONVERTED = "converted" # read, then changed (a unit, edges → a box width) 

28GENERATED = "generated" # the data had none, so the app made a stand-in 

29COMPUTED = "computed" # derived by the app (measures, runs, angles …) 

30YOURS = "yours" # carried through under its own name 

31KINDS = (MAPPED, COMPOSITE, CONVERTED, GENERATED, COMPUTED, YOURS) 

32 

33#: Columns the app derives (measures.py, preprocessing.py, alignment.py, 

34#: data.normalize_*) when the data did not bring them. A column the dataset 

35#: *did* bring under one of these names is the user's: its map entry wins. 

36COMPUTED_COLUMNS: frozenset[str] = frozenset( 

37 { 

38 # data.normalize_fixations / harmonize_frames 

39 "order_in_trial", 

40 "order_in_screen", 

41 "right_to_left", 

42 # measures.compute_per_word_measures 

43 "first_fixation_ms", 

44 "first_pass_gaze_duration_ms", 

45 "regression_path_duration_ms", 

46 "total_fixation_duration_ms", 

47 "gaze_duration_ms", 

48 "n_fixations", 

49 "skip_flag", 

50 "regression_in_flag", 

51 "regression_out_flag", 

52 "first_fix_x", 

53 "first_fix_y", 

54 "initial_landing_position", 

55 "initial_landing_distance", 

56 "number_of_regressions_in", 

57 "second_pass_duration_ms", 

58 "single_fixation_duration_ms", 

59 # measures.enrich_fixations / materialize_runs 

60 "saccade_amplitude", 

61 "angle_incoming", 

62 "angle_outgoing", 

63 "progression", 

64 "is_regression", 

65 "run", 

66 "linerun", 

67 "word_runid", 

68 "word_run", 

69 "word_run_fix", 

70 "nrun", 

71 "reread", 

72 # preprocessing.preprocess_fixations / merge_short_fixations 

73 "excluded", 

74 "excluded_reason", 

75 "blink_before", 

76 "blink_after", 

77 "original_duration_ms", 

78 # alignment.correct 

79 "y_original", 

80 "y_correction", 

81 "alignment_agreement", 

82 } 

83) 

84 

85#: What marks a column the app made, wherever a column is named on screen. 

86#: Text, not an icon: option labels cannot carry a Material icon, and literal 

87#: emoji are kept out of chrome (`tests/test_icons.py`). 

88COMPUTED_SUFFIX = " (computed)" 

89 

90#: Where the open dataset's map lives in session state (stashed by 

91#: `app._stash_active_mapping`), one payload per table. 

92ACTIVE_COLUMN_NAMES_KEY = "_active_column_names" 

93 

94#: Curated labels for columns the app makes. A reading measure's comes from 

95#: `data.READING_MEASURE_FIELDS` (its full name) — see `canonical_label`. 

96_CANONICAL_LABELS: dict[str, str] = { 

97 "order_in_trial": "Fixation order", 

98 "order_in_screen": "Fixation order on screen", 

99 "fixation_id": "Fixation #", 

100 "timestamp_ms": "Time (ms)", 

101 SAMPLE_INDEX: "Sample #", 

102 "participant_id": "Participant", 

103 "text_id": "Text", 

104 "line_idx": "Line", 

105 "text": "Word", 

106 "word_id": "Word #", 

107 "x": "X", 

108 "y": "Y", 

109 "saccade_amplitude": "Saccade amplitude (px)", 

110 "angle_incoming": "Incoming angle (°)", 

111 "angle_outgoing": "Outgoing angle (°)", 

112 "progression": "Progression", 

113 "is_regression": "Regression", 

114 "right_to_left": "Right to left", 

115 "first_fix_x": "First fixation X", 

116 "first_fix_y": "First fixation Y", 

117 "gaze_duration_ms": "Gaze duration (ms)", 

118 "run": "Run", 

119 "linerun": "Line run", 

120 "word_runid": "Word run id", 

121 "word_run": "Word run", 

122 "word_run_fix": "Fixation in word run", 

123 "nrun": "Runs on word", 

124 "reread": "Reread", 

125 "excluded": "Excluded", 

126 "excluded_reason": "Excluded because", 

127 "blink_before": "Blink before", 

128 "blink_after": "Blink after", 

129 "original_duration_ms": "Original duration (ms)", 

130 "y_original": "Y before correction", 

131 "y_correction": "Y correction", 

132 "alignment_agreement": "Line agreement", 

133} 

134 

135 

136#: #374 F5: a mapped **role** — a field the add-dataset wizard asks for — is 

137#: named by its role wherever the app names a field (chips, trial filters, 

138#: figure text), with the dataset's own column in a tooltip 

139#: (:meth:`ColumnNames.source_tooltip`). The dataset's other columns keep their 

140#: own names. The mapping screens themselves still offer the source columns 

141#: (:meth:`ColumnNames.label`). 

142ROLE_LABELS: dict[str, str] = { 

143 "participant_id": "Participant", 

144 "trial_id": "Trial", 

145 "unique_trial_id": "Trial", 

146 "text_id": "Text", 

147 "unique_text_id": "Text", 

148 "screen_id": "Screen", 

149 "word_id": "Word #", 

150 "text": "Word", 

151 "line_idx": "Line", 

152 "x": "X", 

153 "y": "Y", 

154 "width": "Width", 

155 "height": "Height", 

156 "duration_ms": "Duration (ms)", 

157 "timestamp_ms": "Time (ms)", 

158 "fixation_id": "Fixation #", 

159 "screen_fixation_id": "Fixation # on screen", 

160 "canvas_width": "Screen width", 

161 "canvas_height": "Screen height", 

162} 

163 

164#: #374 F5: one-sentence descriptions of the bundled OneStop demo's own columns, 

165#: shown as a tooltip beside their (unchanged) names. Only what the repo's docs 

166#: state (docs/onestop.md, docs/glossary.md); a column not listed gets none. 

167DEMO_COLUMN_NOTES: dict[str, str] = { 

168 "difficulty_level": "The paragraph's version: Adv (Advanced) or Ele (Elementary).", 

169 "question_preview": "True when the question was shown before the paragraph " 

170 "(information seeking).", 

171 "repeated_reading_trial": "True when the paragraph is read for the second time.", 

172 "is_correct": "Whether the comprehension question was answered correctly.", 

173 "selected_answer": "The answer the participant chose.", 

174 "question": "The comprehension question of the trial.", 

175 "article_title": "The title of the article the paragraph comes from.", 

176 "TRIAL_INDEX": "The trial's position in the participant's session.", 

177 "is_in_aspan": "True for the words that answer the trial's question " 

178 "(the answer span).", 

179 "is_in_dspan": "True for the words of the distractor span.", 

180} 

181 

182#: A short name for a highlight column in the figure's key (#374 F6). 

183HIGHLIGHT_NOTES: dict[str, str] = { 

184 "is_in_aspan": "answer span", 

185 "is_in_dspan": "distractor span", 

186} 

187 

188 

189#: Canonical columns that can carry a copy of a partner's values — see 

190#: `ColumnNames.aliases`. 

191_ALIAS_PAIRS = (("trial_id", "unique_trial_id"), ("text_id", "unique_text_id")) 

192 

193#: The note a time column read in another unit carries ("FPOGD, in ms"). A 

194#: figure drops it: its hover writes the unit after the value. 

195IN_MS = ", in ms" 

196 

197#: The id columns a derived table (a summary, a saccade table) shares with the 

198#: dataset's own — the only ones it may carry under the file's names. 

199IDENTITY_COLUMNS = frozenset( 

200 { 

201 "participant_id", 

202 "trial_id", 

203 "unique_trial_id", 

204 "text_id", 

205 "unique_text_id", 

206 "screen_id", 

207 "word_id", 

208 } 

209) 

210 

211 

212def canonical_label(column) -> str: 

213 """A readable label for a column the app made (a measure, a run, an angle …).""" 

214 from .data import READING_MEASURE_FIELDS 

215 

216 column = str(column) 

217 for _key, canonical, _short, full, *_ in READING_MEASURE_FIELDS: 

218 if canonical == column: 

219 return str(full) 

220 if column in _CANONICAL_LABELS: 

221 return _CANONICAL_LABELS[column] 

222 text = column.replace("_", " ").strip() 

223 return text[:1].upper() + text[1:] 

224 

225 

226@dataclass(frozen=True) 

227class SourceName: 

228 """What one canonical column was read from: its source column(s) and how.""" 

229 

230 sources: tuple[str, ...] 

231 kind: str = MAPPED 

232 note: str = "" 

233 

234 @property 

235 def display(self) -> str: 

236 return " + ".join(self.sources) 

237 

238 

239@dataclass(frozen=True) 

240class ColumnNames: 

241 """One table's canonical column → :class:`SourceName` record.""" 

242 

243 entries: Mapping[str, SourceName] = field(default_factory=dict) 

244 

245 def source(self, column) -> SourceName | None: 

246 return self.entries.get(str(column)) 

247 

248 def kind_of(self, column) -> str: 

249 entry = self.source(column) 

250 if entry is not None: 

251 return entry.kind 

252 return COMPUTED if str(column) in COMPUTED_COLUMNS else YOURS 

253 

254 def display(self, column) -> str: 

255 """The name to show for ``column`` — the user's when there is one.""" 

256 entry = self.source(column) 

257 return entry.display if entry is not None and entry.sources else str(column) 

258 

259 def to_canonical(self, name) -> str: 

260 """The canonical column a user's ``name`` stands for (else ``name``).""" 

261 name = str(name) 

262 for column, entry in self.entries.items(): 

263 if entry.sources == (name,) and entry.kind in (MAPPED, CONVERTED): 

264 return column 

265 return name 

266 

267 def through(self, earlier: ColumnNames) -> ColumnNames: 

268 """This map with each source renamed by ``earlier``. 

269 

270 ✏️ Edit dataset maps fields onto the stored frame's *canonical* columns, 

271 so a map built from that edit names canonical columns as its sources; 

272 read through the dataset's earlier map they are the user's names again. 

273 A source that was generated or converted stays so. Columns this map does 

274 not rebuild keep their earlier record. 

275 """ 

276 out: dict[str, SourceName] = {} 

277 for column, entry in self.entries.items(): 

278 sources: list[str] = [] 

279 kind, note = entry.kind, entry.note 

280 for source in entry.sources: 

281 prior = earlier.entries.get(source) 

282 if prior is None: 

283 sources.append(source) 

284 continue 

285 sources.extend(prior.sources) 

286 if kind == MAPPED and prior.kind != MAPPED: 

287 kind, note = prior.kind, prior.note 

288 out[column] = SourceName(tuple(sources), kind, note) 

289 for column, entry in earlier.entries.items(): 

290 out.setdefault(column, entry) 

291 return ColumnNames(out) 

292 

293 def label(self, column) -> str: 

294 """What a person is shown for ``column`` (DATA-66 phase 2). 

295 

296 The user's own name when the column was read from their files; a 

297 curated label marked :data:`COMPUTED_SUFFIX` when the app made it; else 

298 the column's own name. 

299 """ 

300 entry = self.source(column) 

301 kind = self.kind_of(column) 

302 if ( 

303 entry is not None 

304 and entry.sources 

305 and kind in (MAPPED, COMPOSITE, CONVERTED) 

306 ): 

307 # A converted column says what it holds now: a box width is the 

308 # difference of two edges, not their sum, and a duration read in 

309 # seconds is in ms. 

310 if kind == CONVERTED and entry.note: 

311 return entry.note 

312 return entry.display 

313 if kind == GENERATED and str(column) == "timestamp_ms": 

314 # #374: no onset in the data — the stand-in is the order, not ms. 

315 return "Fixation order" + COMPUTED_SUFFIX 

316 if kind in (COMPUTED, GENERATED): 

317 return canonical_label(column) + COMPUTED_SUFFIX 

318 return str(column) 

319 

320 def field_label(self, column) -> str: 

321 """What ``column`` is called where the app names a *field* — a chip, a 

322 trial filter, a picker on the rail (#374 F5). 

323 

324 A role the dataset mapped (:data:`ROLE_LABELS`) is named by its role, 

325 not by the source column; everything else as :meth:`label`. 

326 """ 

327 column = str(column) 

328 if column in ROLE_LABELS and self.kind_of(column) not in ( 

329 COMPUTED, 

330 GENERATED, 

331 ): 

332 return ROLE_LABELS[column] 

333 return self.label(column) 

334 

335 def source_tooltip(self, column) -> str: 

336 """``"from RECORDING_SESSION_LABEL"`` for a role :meth:`field_label` 

337 names by its role, else ``""`` (#374 F5).""" 

338 column = str(column) 

339 entry = self.source(column) 

340 if column not in ROLE_LABELS or entry is None or not entry.sources: 

341 return "" 

342 if entry.kind == CONVERTED and entry.note: 

343 return f"from {entry.note}" 

344 if entry.kind not in (MAPPED, COMPOSITE): 

345 return "" 

346 return f"from {entry.display}" 

347 

348 def figure_labels(self, columns: Iterable) -> dict[str, str]: 

349 """``{column: label}`` for a figure's text (`FigureSettings.column_labels`). 

350 

351 Every column the dataset brought, under its own name, except a mapped 

352 role (:data:`ROLE_LABELS`); those and the columns the app made are left 

353 out, so a figure keeps its short labels for them 

354 ("Fixation #", "FFD") rather than writing "(computed)" into a hover. 

355 Bookkeeping columns (`data.INTERNAL_COLUMNS`) are no figure's text. A 

356 column converted from one source (a duration read in seconds) is named 

357 by that source: the figure writes its unit after the value, and the 

358 note's "…, in ms" would say it twice. 

359 """ 

360 from .data import INTERNAL_COLUMNS 

361 

362 out: dict[str, str] = {} 

363 for column in columns: 

364 # #374 F5/F7: a role is the figure's own plain word ("Duration", 

365 # "Word #"), not the source column. 

366 if ( 

367 column in INTERNAL_COLUMNS 

368 or str(column) in ROLE_LABELS 

369 or self.kind_of(column) in (COMPUTED, GENERATED) 

370 ): 

371 continue 

372 entry = self.source(column) 

373 unit_conversion = ( 

374 entry is not None 

375 and entry.kind == CONVERTED 

376 and entry.note.endswith(IN_MS) 

377 ) 

378 out[str(column)] = ( 

379 entry.note.removesuffix(IN_MS) 

380 if unit_conversion 

381 else self.label(column) 

382 ) 

383 return out 

384 

385 def merged(self, other: ColumnNames) -> ColumnNames: 

386 """Both tables' entries, this map's winning where both name a column.""" 

387 return ColumnNames({**dict(other.entries), **dict(self.entries)}) 

388 

389 def option_labels( 

390 self, options, extra: Mapping | None = None, *, roles: bool = False 

391 ) -> dict[str, str]: 

392 """``{option: label}`` for a picker; ``extra`` labels synthetic options 

393 (``"(uniform)"``, ``"line"``). A label two options share gets the 

394 internal name added, so a picker never shows two identical rows. 

395 ``roles=True`` names a mapped role by its role (:meth:`field_label`) — 

396 for the rail's pickers, not the mapping screens.""" 

397 extra = dict(extra or {}) 

398 name = self.field_label if roles else self.label 

399 labels = {o: extra.get(o) or name(o) for o in options} 

400 counts: dict[str, int] = {} 

401 for value in labels.values(): 

402 counts[value] = counts.get(value, 0) + 1 

403 return { 

404 o: f"{label} · {o}" if counts[label] > 1 and label != str(o) else label 

405 for o, label in labels.items() 

406 } 

407 

408 def sort_options(self, options, first=()) -> list: 

409 """The user's columns first, the app's last; ``first`` stays in front. 

410 

411 Stable within each group, so a curated order survives.""" 

412 options = list(options) 

413 head = [o for o in options if o in first] 

414 rest = [o for o in options if o not in first] 

415 made = (COMPUTED, GENERATED) 

416 return ( 

417 head 

418 + [o for o in rest if self.kind_of(o) not in made] 

419 + [o for o in rest if self.kind_of(o) in made] 

420 ) 

421 

422 def aliases(self, columns: Iterable[str]) -> set[str]: 

423 """Columns in ``columns`` that only repeat a partner from the same source. 

424 

425 `unique_trial_id` mirrors `trial_id` (BUG-58) and `unique_text_id` 

426 usually mirrors `text_id`; when both of a pair were read from one column 

427 of the user's file, a table needs to show it once. 

428 """ 

429 present = {str(c) for c in columns} 

430 hidden: set[str] = set() 

431 for main, alias in _ALIAS_PAIRS: 

432 first, second = self.source(main), self.source(alias) 

433 if ( 

434 main in present 

435 and alias in present 

436 and first is not None 

437 and second is not None 

438 and first.sources 

439 and first.sources == second.sources 

440 ): 

441 hidden.add(alias) 

442 return hidden 

443 

444 def with_rewrites(self, table: str, rewrites: Iterable) -> ColumnNames: 

445 """This map with every column the load rewrote (``data.Rewrite``s for 

446 ``table``) marked converted: a padded id, a shifted word id, positions 

447 filled from word boxes are no longer what the file held under its name, 

448 so the column is labelled "<source><how>" and exported under its 

449 internal name.""" 

450 changed = dict(self.entries) 

451 for rewritten_table, column, how in rewrites or (): 

452 entry = changed.get(column) 

453 if rewritten_table != table or entry is None or entry.kind != MAPPED: 

454 continue 

455 note = " + ".join(entry.sources) + how 

456 changed[column] = SourceName(entry.sources, CONVERTED, note) 

457 return ColumnNames(changed) 

458 

459 def identity(self) -> ColumnNames: 

460 """This map's id columns only (:data:`IDENTITY_COLUMNS`). 

461 

462 For a table the app derives — a summary, a saccade table, a character 

463 grid — whose other columns reuse a canonical name (`n_fixations`, 

464 `duration_ms`, `x`) for a value of their own: only its ids are the 

465 file's.""" 

466 return self.restricted_to(IDENTITY_COLUMNS) 

467 

468 def redundant_aliases(self, frame) -> set[str]: 

469 """:meth:`aliases` that really repeat their partner in ``frame``. 

470 

471 The map says both came from one column; the values decide, since a 

472 padded or suffixed id can part from its copy (BUG-59).""" 

473 partners = {alias: main for main, alias in _ALIAS_PAIRS} 

474 return { 

475 alias 

476 for alias in self.aliases(frame.columns) 

477 if frame[alias].equals(frame[partners[alias]]) 

478 } 

479 

480 def export_headers(self, columns: Iterable) -> dict[str, str]: 

481 """``{column: header}`` for a table written out (DATA-66 phase 3). 

482 

483 A column read from one column of the user's file is written under that 

484 column's name. One built from several (a composite trial id), converted 

485 (a box width from two edges, a duration read in seconds) or made by the 

486 app keeps its canonical name — its values are not what the file held 

487 under any one name — and the bundle's `columns.json` and README say 

488 where it came from. A header that would repeat another column's name 

489 keeps the canonical one. Columns not renamed are left out. 

490 """ 

491 names = [str(c) for c in columns] 

492 taken = set(names) 

493 out: dict[str, str] = {} 

494 for column in names: 

495 entry = self.source(column) 

496 if entry is None or entry.kind != MAPPED or len(entry.sources) != 1: 

497 continue 

498 header = entry.sources[0] 

499 if header == column or header in taken: 

500 continue 

501 taken.add(header) 

502 out[column] = header 

503 return out 

504 

505 def provenance(self, column) -> str: 

506 """Where ``column`` came from, in a sentence fragment (the README's).""" 

507 entry = self.source(column) 

508 kind = self.kind_of(column) 

509 sources = ", ".join(f"`{s}`" for s in entry.sources) if entry else "" 

510 note = entry.note if entry else "" 

511 if kind == MAPPED and sources: 

512 return f"your column {sources}" 

513 if kind == COMPOSITE: 

514 return f"joined from your columns {sources}" 

515 if kind == CONVERTED: 

516 return f"converted from your {sources}" + (f" ({note})" if note else "") 

517 if kind == GENERATED: 

518 return "made by Scanpath Studio" + (f" ({note})" if note else "") 

519 if kind == COMPUTED: 

520 return "computed by Scanpath Studio" 

521 return "your column, under its own name" 

522 

523 def restricted_to(self, columns: Iterable[str]) -> ColumnNames: 

524 """Only the entries for ``columns`` — a frame's actual columns. 

525 

526 After an edit, :meth:`through` keeps every earlier entry the edit did 

527 not rebuild, including one for a column the edit removed (a cleared 

528 reading measure); restricting to the saved frame drops those, so the 

529 record never names a column the dataset no longer has. 

530 """ 

531 keep = {str(c) for c in columns} 

532 return ColumnNames({c: e for c, e in self.entries.items() if c in keep}) 

533 

534 def to_payload(self) -> dict: 

535 """A JSON-safe form, for a `_datasets` entry and the recovery cache.""" 

536 return { 

537 column: {"sources": list(e.sources), "kind": e.kind, "note": e.note} 

538 for column, e in self.entries.items() 

539 } 

540 

541 @classmethod 

542 def from_payload(cls, payload) -> ColumnNames: 

543 """The inverse of :meth:`to_payload`; anything malformed reads as empty.""" 

544 if not isinstance(payload, Mapping): 

545 return EMPTY 

546 entries: dict[str, SourceName] = {} 

547 for column, raw in payload.items(): 

548 if not isinstance(raw, Mapping): 

549 return EMPTY 

550 kind = raw.get("kind", MAPPED) 

551 sources = raw.get("sources") or () 

552 if kind not in KINDS or not isinstance(sources, Iterable): 

553 return EMPTY 

554 entries[str(column)] = SourceName( 

555 tuple(str(s) for s in sources), kind, str(raw.get("note") or "") 

556 ) 

557 return cls(entries) 

558 

559 

560EMPTY = ColumnNames({}) 

561 

562 

563def active(session: Mapping, table: str) -> ColumnNames: 

564 """The open dataset's map for ``table``, read from a session mapping. 

565 

566 Takes the session as an argument so this module stays free of Streamlit; 

567 callers pass ``st.session_state``. 

568 """ 

569 stash = session.get(ACTIVE_COLUMN_NAMES_KEY) or {} 

570 return ColumnNames.from_payload(stash.get(table)) 

571 

572 

573def active_all(session: Mapping) -> ColumnNames: 

574 """The open dataset's map over all its tables, for a label any table can own. 

575 

576 The fixations table's entries win, then the words table's (word-level 

577 fields such as surprisal are carried onto fixations), then raw gaze's. 

578 """ 

579 return across_tables({table: active(session, table) for table in _TABLES}) 

580 

581 

582def table_figure_labels( 

583 maps: Mapping[str, ColumnNames], columns: Mapping[str, Iterable] 

584) -> dict[str, str]: 

585 """`FigureSettings.column_labels` for a figure drawn from several tables. 

586 

587 One flat map cannot hold two names for one column — ``word_id`` is the AOI 

588 table's ``IA_ID`` and the fixation table's interest-area column — so the 

589 merged map's labels (fixations' winning) sit under the plain keys and a 

590 word column the words table names differently also under 

591 ``"words:<column>"``, which a word hover reads first (`plots._table_label`). 

592 Only the tables the figure is drawn from are consulted, so a words-only 

593 figure is labelled by the words table alone. 

594 """ 

595 columns = {table: list(cols) for table, cols in columns.items()} 

596 drawn = {table: names for table, names in maps.items() if table in columns} 

597 out = across_tables(drawn).figure_labels( 

598 [column for cols in columns.values() for column in cols] 

599 ) 

600 words = maps.get("words") 

601 if words is not None and "words" in columns: 

602 for column, label in words.figure_labels(columns["words"]).items(): 

603 if out.get(column) != label: 

604 out[f"words:{column}"] = label 

605 return out 

606 

607 

608def active_figure_labels(session: Mapping, **columns: Iterable) -> dict[str, str]: 

609 """:func:`table_figure_labels` for the open dataset — ``columns`` keyed by 

610 table (``words=…``, ``fixations=…``).""" 

611 return table_figure_labels( 

612 {table: active(session, table) for table in _TABLES}, columns 

613 ) 

614 

615 

616#: The tables a dataset's map covers, in the order their entries win. 

617_TABLES = ("fixations", "words", "raw_gaze") 

618 

619 

620def across_tables(maps: Mapping[str, ColumnNames]) -> ColumnNames: 

621 """One map from a dataset's per-table maps (fixations', then words', then 

622 raw gaze's entries win) — what :func:`active_all` reads from the session, 

623 for a dataset held elsewhere (Compare's B, `SecondaryDataset.column_names`).""" 

624 out = EMPTY 

625 for table in _TABLES: 

626 if table in maps: 

627 out = out.merged(maps[table]) 

628 return out 

629 

630 

631#: Mapped screen fields: schema key → canonical column (`data._copy_screen_fields`). 

632_SCREEN_FIELDS = ( 

633 ("screen_id", "screen_id"), 

634 ("screen_index", "screen_index"), 

635 ("screen_timestamp", "screen_timestamp_ms"), 

636 ("screen_fixation_id", "screen_fixation_id"), 

637 ("canvas_width", "canvas_width"), 

638 ("canvas_height", "canvas_height"), 

639) 

640 

641 

642def _id_entry(value) -> SourceName | None: 

643 """A schema id field — one column, or several joined (a composite id).""" 

644 from .data import trial_mapping_columns 

645 

646 if not value: 

647 return None 

648 columns = [str(c) for c in trial_mapping_columns(value)] 

649 if len(columns) > 1: 

650 return SourceName(tuple(columns), COMPOSITE) 

651 return SourceName((columns[0],), MAPPED) 

652 

653 

654def _timed(column: str) -> SourceName: 

655 """A time column, which normalization converts to ms when its unit says so.""" 

656 from .data import time_unit_ms 

657 

658 if time_unit_ms(column) != 1.0: 

659 return SourceName((column,), CONVERTED, f"{column}, in ms") 

660 return SourceName((column,), MAPPED) 

661 

662 

663def _mapped_or(schema: Mapping, key: str, kind: str, note: str) -> SourceName: 

664 """The schema's column for ``key``, else a stand-in of ``kind``.""" 

665 column = schema.get(key) 

666 return SourceName((str(column),)) if column else SourceName((), kind, note) 

667 

668 

669def _registry(out: dict, registry, present: set, keep: set | None) -> None: 

670 """The optional-field renames normalization applied (`_apply_optional_fields`). 

671 

672 A later entry for the same destination overwrites an earlier one there, so 

673 it does here too; a passthrough (`src == dest`) keeps its own name and needs 

674 no entry. 

675 """ 

676 for src, dest, _kind, _category in registry: 

677 if src not in present or (keep is not None and src not in keep): 

678 continue 

679 if dest != src: 

680 out[dest] = SourceName((src,), MAPPED) 

681 

682 

683def from_schema( 

684 table: str, 

685 schema: Mapping | None, 

686 columns: Iterable[str], 

687 *, 

688 keep_columns: Iterable[str] | None = None, 

689) -> ColumnNames: 

690 """The map ``data.normalize_<table>`` implies for ``schema`` over ``columns``. 

691 

692 ``table`` is ``"words"``, ``"fixations"`` or ``"raw_gaze"``; ``columns`` are 

693 the raw table's; ``keep_columns`` the set normalization was given (``None`` 

694 carries every registry field, as normalization does). 

695 """ 

696 from . import data 

697 

698 schema = dict(schema or {}) 

699 present = {str(c) for c in columns} 

700 keep = None if keep_columns is None else {str(c) for c in keep_columns} 

701 out: dict[str, SourceName] = {} 

702 

703 out["participant_id"] = _id_entry(schema.get("participant")) or SourceName( 

704 (), GENERATED, "one participant for the whole table" 

705 ) 

706 if trial := _id_entry(schema.get("trial")): 

707 out["trial_id"] = out["unique_trial_id"] = trial 

708 if table != "raw_gaze" and "unique_paragraph_id" in present: 

709 out["text_id"] = out["unique_text_id"] = SourceName(("unique_paragraph_id",)) 

710 elif text_id := _id_entry(schema.get("text_id")): 

711 # A remap fills `unique_text_id` from the mapped Text ID 

712 # (`data.remap_normalized_frame`); a first load has no such column, and 

713 # an entry for an absent column names nothing. A `unique_text_id` the 

714 # file itself has is the user's own column, under its own name. 

715 out["text_id"] = text_id 

716 if "unique_text_id" not in present: 

717 out["unique_text_id"] = text_id 

718 else: 

719 out["text_id"] = SourceName((), GENERATED, "the trial id") 

720 for key, canonical in _SCREEN_FIELDS: 

721 if schema.get(key): 

722 out[canonical] = SourceName((str(schema[key]),)) 

723 

724 if table == "words": 

725 out["word_id"] = _mapped_or(schema, "word_id", GENERATED, "the row order") 

726 out["text"] = _mapped_or(schema, "text", GENERATED, "w0, w1 … from the word id") 

727 out["line_idx"] = _mapped_or(schema, "line", GENERATED, "1 — one line") 

728 if all(schema.get(k) for k in ("x", "y", "width", "height")): 

729 for key in ("x", "y", "width", "height"): 

730 out[key] = SourceName((str(schema[key]),)) 

731 elif all(schema.get(k) for k in ("left", "right", "top", "bottom")): 

732 left, right = str(schema["left"]), str(schema["right"]) 

733 top, bottom = str(schema["top"]), str(schema["bottom"]) 

734 out["x"], out["y"] = SourceName((left,)), SourceName((top,)) 

735 out["width"] = SourceName((right, left), CONVERTED, f"{right} − {left}") 

736 out["height"] = SourceName((bottom, top), CONVERTED, f"{bottom} − {top}") 

737 _registry(out, data.WORD_OPTIONAL_FIELDS, present, keep) 

738 # AN-32: a measure the schema names decides it, after the passthrough. 

739 for key, canonical, *_ in data.READING_MEASURE_FIELDS: 

740 if key not in schema: 

741 continue 

742 column = schema.get(key) 

743 if column and column in present: 

744 out[canonical] = SourceName((str(column),)) 

745 else: 

746 out.pop(canonical, None) 

747 elif table == "fixations": 

748 for coord in ("x", "y"): 

749 out[coord] = _mapped_or( 

750 schema, coord, COMPUTED, "the fixated word's box center" 

751 ) 

752 if schema.get("duration"): 

753 out["duration_ms"] = _timed(str(schema["duration"])) 

754 out["timestamp_ms"] = ( 

755 _timed(str(schema["timestamp"])) 

756 if schema.get("timestamp") 

757 else SourceName((), GENERATED, "the fixation's order in its trial") 

758 ) 

759 out["fixation_id"] = _mapped_or( 

760 schema, "fixation_id", GENERATED, "1, 2, … per trial" 

761 ) 

762 out["word_id"] = _mapped_or( 

763 schema, "word_id", COMPUTED, "assigned from the word boxes" 

764 ) 

765 _registry(out, data.FIX_OPTIONAL_FIELDS, present, keep) 

766 else: # raw gaze 

767 for key in ("x", "y", "text", "word_id"): 

768 if schema.get(key): 

769 out[key] = SourceName((str(schema[key]),)) 

770 if schema.get("timestamp"): 

771 out["timestamp_ms"] = _timed(str(schema["timestamp"])) 

772 else: 

773 # No clock: the samples are numbered, never given a time. 

774 out[SAMPLE_INDEX] = SourceName( 

775 (), GENERATED, "the sample's order in its trial (no time mapped)" 

776 ) 

777 return ColumnNames(out) 

778 

779 

780def for_tables( 

781 schemas: Mapping[str, Mapping | None], 

782 frames: Mapping[str, object], 

783 keeps: Mapping[str, Iterable[str] | None] | None = None, 

784 rewrites: Iterable | None = None, 

785) -> dict[str, dict]: 

786 """``{table: payload}`` for every table with a schema and a raw frame. 

787 

788 ``rewrites`` are the ``data.Rewrite``s the load made 

789 (:meth:`ColumnNames.with_rewrites`).""" 

790 keeps = keeps or {} 

791 rewrites = tuple(rewrites or ()) 

792 out: dict[str, dict] = {} 

793 for table, schema in schemas.items(): 

794 columns = getattr(frames.get(table), "columns", None) 

795 if not schema or columns is None or len(columns) == 0: 

796 continue 

797 names = from_schema(table, schema, columns, keep_columns=keeps.get(table)) 

798 out[table] = names.with_rewrites(table, rewrites).to_payload() 

799 return out 

800 

801 

802# --- Phase 3: what a written table calls its columns ------------------------- 

803 

804#: The `columns.json` format a bundle writes beside its tables. 

805COLUMNS_FILE_SCHEMA = 1 

806 

807 

808def _written_plan(frame, names: ColumnNames) -> tuple[set[str], dict[str, str]]: 

809 """``(left out, renamed)`` for writing ``frame``: the `unique_*` aliases that 

810 only repeat their partner, and each remaining column's header.""" 

811 hidden = names.redundant_aliases(frame) 

812 kept = [c for c in frame.columns if c not in hidden] 

813 return hidden, names.export_headers(kept) 

814 

815 

816def as_written(frame, names: ColumnNames | None, hidden: set[str] | None = None): 

817 """``frame`` as a bundle writes it: each column the user's file named under 

818 that name (`ColumnNames.export_headers`), and a `unique_*` alias that only 

819 repeats its partner left out. The same object when nothing changes. 

820 

821 ``hidden`` is the aliases to leave out, decided once for the whole table 

822 (:meth:`ColumnNames.redundant_aliases`) so every per-trial file of it has 

823 the same columns; without it, ``frame`` decides.""" 

824 if names is None or not names.entries: 

825 return frame 

826 if hidden is None: 

827 hidden, headers = _written_plan(frame, names) 

828 else: 

829 hidden = {c for c in hidden if c in frame.columns} 

830 headers = names.export_headers([c for c in frame.columns if c not in hidden]) 

831 if hidden: 

832 frame = frame.drop(columns=sorted(hidden)) 

833 return frame.rename(columns=headers) if headers else frame 

834 

835 

836def written_columns(frame, names: ColumnNames | None) -> list[dict]: 

837 """What :func:`as_written` makes of ``frame``'s mapped columns, one row each: 

838 the header written, the internal (canonical) name, its kind, its sources and 

839 note. Columns the map does not record are written under their own names and 

840 are not listed.""" 

841 if names is None or not names.entries: 

842 return [] 

843 hidden, headers = _written_plan(frame, names) 

844 rows = [] 

845 for column in frame.columns: 

846 entry = names.source(column) 

847 if column in hidden or entry is None: 

848 continue 

849 rows.append( 

850 { 

851 "column": headers.get(column, column), 

852 "canonical": column, 

853 "kind": entry.kind, 

854 "sources": list(entry.sources), 

855 "note": entry.note, 

856 } 

857 ) 

858 return rows 

859 

860 

861def columns_manifest(tables: Mapping[str, list[dict]]) -> dict: 

862 """The `columns.json` a bundle carries: ``{table: written_columns(…)}``, so a 

863 script can map every file back to the internal names.""" 

864 return { 

865 "schema": COLUMNS_FILE_SCHEMA, 

866 "tables": {table: rows for table, rows in tables.items() if rows}, 

867 } 

868 

869 

870def dictionary_lines( 

871 tables: Mapping[str, list[dict]], names: Mapping[str, ColumnNames] 

872) -> list[str]: 

873 """The README's data dictionary: for each table's :func:`written_columns`, 

874 the header written and where it came from, in markdown.""" 

875 titles = { 

876 "fixations": "Fixations", 

877 "words": "Words (interest areas)", 

878 "raw_gaze": "Raw gaze", 

879 } 

880 lines: list[str] = [] 

881 for table, rows in tables.items(): 

882 if not rows: 

883 continue 

884 lines += ["", f"### {titles.get(table, table)}"] 

885 for row in rows: 

886 written, column = row["column"], row["canonical"] 

887 internal = "" if written == column else f" (internally `{column}`)" 

888 lines.append(f"- `{written}`{internal}: {names[table].provenance(column)}") 

889 return lines 

890 

891 

892def source_schema( 

893 schema: Mapping | None, names: ColumnNames 

894) -> tuple[dict | None, tuple[str, ...]]: 

895 """``schema`` restated in the dataset's own files' column names. 

896 

897 ✏️ Edit dataset maps fields onto the stored frame's *canonical* columns 

898 (``{"trial": "trial_id", "x": "x"}``), which say nothing to a script that 

899 reads the original files. Read through ``names`` — the map from before 

900 that edit — each becomes the column(s) it was read from, so the result can 

901 be handed to ``api.load_scanpath_data`` over those files (Share → Code). A 

902 column the app *made* (a generated stand-in, a computed value) maps to 

903 nothing, which makes the loader make it again; a box stored as 

904 ``x/y/width/height`` but read from edges is restated as those edges. 

905 

906 Returns ``(schema, unresolved)``: the fields that could not be traced back, 

907 left as they were, for the caller to name. 

908 """ 

909 from .data import trial_mapping_columns 

910 

911 if schema is None: 

912 return None, () 

913 out: dict = {} 

914 unresolved: list[str] = [] 

915 for key, value in schema.items(): 

916 if not value: 

917 out[key] = value 

918 continue 

919 columns: list[str] = [] 

920 made = traced = False 

921 for column in trial_mapping_columns(value): 

922 entry = names.source(column) 

923 if entry is None: 

924 columns.append(str(column)) 

925 elif entry.kind in (GENERATED, COMPUTED) or not entry.sources: 

926 made = True 

927 elif entry.kind == CONVERTED and len(entry.sources) != 1: 

928 traced = True # right − left: only the edges say it 

929 else: 

930 columns.extend(entry.sources) 

931 if traced: 

932 out[key] = value 

933 unresolved.append(str(key)) 

934 elif made: 

935 out[key] = None 

936 elif isinstance(value, str) and len(columns) == 1: 

937 out[key] = columns[0] 

938 else: 

939 out[key] = columns 

940 # A box read from edges is stored as x/y/width/height, its width and height 

941 # computed as right − left: restate it as the edges it came from. 

942 sizes = [names.source(schema.get(side) or "") for side in ("width", "height")] 

943 if ( 

944 {"width", "height"} <= set(unresolved) 

945 and all(e is not None and len(e.sources) == 2 for e in sizes) 

946 and isinstance(out.get("x"), str) 

947 and isinstance(out.get("y"), str) 

948 ): 

949 out.update( 

950 left=out["x"], 

951 top=out["y"], 

952 right=sizes[0].sources[0], 

953 bottom=sizes[1].sources[0], 

954 x=None, 

955 y=None, 

956 width=None, 

957 height=None, 

958 ) 

959 unresolved = [k for k in unresolved if k not in ("width", "height")] 

960 return out, tuple(unresolved) 

961 

962 

963def stored_source_recipe(stored: Mapping) -> dict: 

964 """A stored dataset's ``source_recipe`` — how a script loads its files. 

965 

966 Written when the dataset is added (`wizard._source_recipe`) and kept 

967 current by ✏️ Edit dataset; read by Share → Code 

968 (`code_snippet.upload_source`). A dataset stored before the recipe existed 

969 gets one read off its stored mapping through its column-name map, which is 

970 right unless one of its files used a canonical column name for a different 

971 field. 

972 """ 

973 recipe = stored.get("source_recipe") 

974 if isinstance(recipe, Mapping): 

975 return dict(recipe) 

976 names = stored.get("column_names") or {} 

977 schemas: dict = {} 

978 unresolved: dict = {} 

979 for table, schema in (stored.get("schemas") or {}).items(): 

980 schemas[table], missing = source_schema( 

981 schema, ColumnNames.from_payload(names.get(table)) 

982 ) 

983 if missing: 

984 unresolved[table] = list(missing) 

985 return {"schemas": schemas, "unresolved": unresolved} 

986 

987 

988# --- Phase 4: a frame that carries its own names ------------------------------ 

989 

990#: The `DataFrame.attrs` key a frame under the dataset's own names carries its 

991#: map in (`attach`). pandas 3 keeps `attrs` through filtering, `loc`, `copy`, 

992#: `assign` and `groupby`, so a script can slice the frame it loaded and hand it 

993#: back to the API. `merge` and `concat` keep them only when every input carries 

994#: the same `attrs`: a frame joined with a table of the user's own loses its map, 

995#: and `api._require_normalized` says how to get it back. 

996ATTRS_KEY = "scanpath_studio.columns" 

997 

998 

999def attach(frame, table: str, names: ColumnNames | None): 

1000 """``frame`` under the dataset's own names (:func:`as_written`), carrying 

1001 its map in ``attrs`` so :func:`to_canonical_frame` can undo it. A frame 

1002 with no names to apply comes back as it was.""" 

1003 if frame is None or names is None or not names.entries: 

1004 return frame 

1005 hidden, headers = _written_plan(frame, names) 

1006 partners = {alias: main for main, alias in _ALIAS_PAIRS} 

1007 out = as_written(frame, names, hidden) 

1008 # One JSON string, not a nested dict: pandas deep-copies `attrs` on every 

1009 # operation, and walking a dict of dicts made a script's per-trial loop 

1010 # over a corpus 2-3x slower; a string is copied for nothing. 

1011 record = { 

1012 "table": table, 

1013 "names": names.to_payload(), 

1014 "renamed": {header: column for column, header in headers.items()}, 

1015 "aliases": {alias: partners[alias] for alias in hidden}, 

1016 } 

1017 out.attrs = {**frame.attrs, ATTRS_KEY: json.dumps(record)} 

1018 return out 

1019 

1020 

1021def _record(frame) -> dict | None: 

1022 """The record :func:`attach` left on ``frame``, else ``None``.""" 

1023 raw = getattr(frame, "attrs", {}).get(ATTRS_KEY) 

1024 if not isinstance(raw, str): 

1025 return None 

1026 try: 

1027 record = json.loads(raw) 

1028 except ValueError: 

1029 return None 

1030 return record if isinstance(record, dict) else None 

1031 

1032 

1033def frame_names(frame) -> tuple[str, ColumnNames] | None: 

1034 """``(table, map)`` a frame under the dataset's own names carries, else 

1035 ``None`` (a canonical frame, or any other table).""" 

1036 record = _record(frame) 

1037 if record is None: 

1038 return None 

1039 return str(record.get("table", "")), ColumnNames.from_payload(record.get("names")) 

1040 

1041 

1042def to_canonical_frame(frame): 

1043 """A frame :func:`attach` named, back under the canonical names it is 

1044 processed in — the inverse every API function applies on entry. Any other 

1045 frame is returned unchanged.""" 

1046 record = _record(frame) 

1047 if record is None: 

1048 return frame 

1049 renamed = { 

1050 header: column 

1051 for header, column in (record.get("renamed") or {}).items() 

1052 if header in frame.columns and column not in frame.columns 

1053 } 

1054 out = frame.rename(columns=renamed) if renamed else frame.copy(deep=False) 

1055 for alias, main in (record.get("aliases") or {}).items(): 

1056 if alias not in out.columns and main in out.columns: 

1057 out[alias] = out[main] 

1058 out.attrs = {k: v for k, v in frame.attrs.items() if k != ATTRS_KEY} 

1059 return out