Coverage for scanpath_studio/column_names.py: 97%
436 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""DATA-66: the dataset's own column names, behind the canonical ones.
3Normalization (`data.normalize_*`) rebuilds every table under canonical names —
4`duration_ms`, `total_fixation_duration_ms` — and, until this module, kept no
5record of what each was called in the user's files. A :class:`ColumnNames` is
6that record, one per table of a dataset: for each canonical column, the source
7column(s) it was read from and how (`kind`). :func:`from_schema` builds it from
8the same schema, registries and raw columns normalization used, so the two
9cannot disagree. Pure — no Streamlit, no I/O.
11The canonical names stay the internal contract and every wire format's values;
12this map is what a person is shown, what an export writes and what the API
13accepts (DATA-66 phases 2–4).
14"""
16from __future__ import annotations
18import json
19from collections.abc import Iterable, Mapping
20from dataclasses import dataclass, field
22from .constants import SAMPLE_INDEX
24#: How a canonical column came to be.
25MAPPED = "mapped" # read from one source column, unchanged
26COMPOSITE = "composite" # several source columns joined (a composite id)
27CONVERTED = "converted" # read, then changed (a unit, edges → a box width)
28GENERATED = "generated" # the data had none, so the app made a stand-in
29COMPUTED = "computed" # derived by the app (measures, runs, angles …)
30YOURS = "yours" # carried through under its own name
31KINDS = (MAPPED, COMPOSITE, CONVERTED, GENERATED, COMPUTED, YOURS)
33#: Columns the app derives (measures.py, preprocessing.py, alignment.py,
34#: data.normalize_*) when the data did not bring them. A column the dataset
35#: *did* bring under one of these names is the user's: its map entry wins.
36COMPUTED_COLUMNS: frozenset[str] = frozenset(
37 {
38 # data.normalize_fixations / harmonize_frames
39 "order_in_trial",
40 "order_in_screen",
41 "right_to_left",
42 # measures.compute_per_word_measures
43 "first_fixation_ms",
44 "first_pass_gaze_duration_ms",
45 "regression_path_duration_ms",
46 "total_fixation_duration_ms",
47 "gaze_duration_ms",
48 "n_fixations",
49 "skip_flag",
50 "regression_in_flag",
51 "regression_out_flag",
52 "first_fix_x",
53 "first_fix_y",
54 "initial_landing_position",
55 "initial_landing_distance",
56 "number_of_regressions_in",
57 "second_pass_duration_ms",
58 "single_fixation_duration_ms",
59 # measures.enrich_fixations / materialize_runs
60 "saccade_amplitude",
61 "angle_incoming",
62 "angle_outgoing",
63 "progression",
64 "is_regression",
65 "run",
66 "linerun",
67 "word_runid",
68 "word_run",
69 "word_run_fix",
70 "nrun",
71 "reread",
72 # preprocessing.preprocess_fixations / merge_short_fixations
73 "excluded",
74 "excluded_reason",
75 "blink_before",
76 "blink_after",
77 "original_duration_ms",
78 # alignment.correct
79 "y_original",
80 "y_correction",
81 "alignment_agreement",
82 }
83)
85#: What marks a column the app made, wherever a column is named on screen.
86#: Text, not an icon: option labels cannot carry a Material icon, and literal
87#: emoji are kept out of chrome (`tests/test_icons.py`).
88COMPUTED_SUFFIX = " (computed)"
90#: Where the open dataset's map lives in session state (stashed by
91#: `app._stash_active_mapping`), one payload per table.
92ACTIVE_COLUMN_NAMES_KEY = "_active_column_names"
94#: Curated labels for columns the app makes. A reading measure's comes from
95#: `data.READING_MEASURE_FIELDS` (its full name) — see `canonical_label`.
96_CANONICAL_LABELS: dict[str, str] = {
97 "order_in_trial": "Fixation order",
98 "order_in_screen": "Fixation order on screen",
99 "fixation_id": "Fixation #",
100 "timestamp_ms": "Time (ms)",
101 SAMPLE_INDEX: "Sample #",
102 "participant_id": "Participant",
103 "text_id": "Text",
104 "line_idx": "Line",
105 "text": "Word",
106 "word_id": "Word #",
107 "x": "X",
108 "y": "Y",
109 "saccade_amplitude": "Saccade amplitude (px)",
110 "angle_incoming": "Incoming angle (°)",
111 "angle_outgoing": "Outgoing angle (°)",
112 "progression": "Progression",
113 "is_regression": "Regression",
114 "right_to_left": "Right to left",
115 "first_fix_x": "First fixation X",
116 "first_fix_y": "First fixation Y",
117 "gaze_duration_ms": "Gaze duration (ms)",
118 "run": "Run",
119 "linerun": "Line run",
120 "word_runid": "Word run id",
121 "word_run": "Word run",
122 "word_run_fix": "Fixation in word run",
123 "nrun": "Runs on word",
124 "reread": "Reread",
125 "excluded": "Excluded",
126 "excluded_reason": "Excluded because",
127 "blink_before": "Blink before",
128 "blink_after": "Blink after",
129 "original_duration_ms": "Original duration (ms)",
130 "y_original": "Y before correction",
131 "y_correction": "Y correction",
132 "alignment_agreement": "Line agreement",
133}
136#: #374 F5: a mapped **role** — a field the add-dataset wizard asks for — is
137#: named by its role wherever the app names a field (chips, trial filters,
138#: figure text), with the dataset's own column in a tooltip
139#: (:meth:`ColumnNames.source_tooltip`). The dataset's other columns keep their
140#: own names. The mapping screens themselves still offer the source columns
141#: (:meth:`ColumnNames.label`).
142ROLE_LABELS: dict[str, str] = {
143 "participant_id": "Participant",
144 "trial_id": "Trial",
145 "unique_trial_id": "Trial",
146 "text_id": "Text",
147 "unique_text_id": "Text",
148 "screen_id": "Screen",
149 "word_id": "Word #",
150 "text": "Word",
151 "line_idx": "Line",
152 "x": "X",
153 "y": "Y",
154 "width": "Width",
155 "height": "Height",
156 "duration_ms": "Duration (ms)",
157 "timestamp_ms": "Time (ms)",
158 "fixation_id": "Fixation #",
159 "screen_fixation_id": "Fixation # on screen",
160 "canvas_width": "Screen width",
161 "canvas_height": "Screen height",
162}
164#: #374 F5: one-sentence descriptions of the bundled OneStop demo's own columns,
165#: shown as a tooltip beside their (unchanged) names. Only what the repo's docs
166#: state (docs/onestop.md, docs/glossary.md); a column not listed gets none.
167DEMO_COLUMN_NOTES: dict[str, str] = {
168 "difficulty_level": "The paragraph's version: Adv (Advanced) or Ele (Elementary).",
169 "question_preview": "True when the question was shown before the paragraph "
170 "(information seeking).",
171 "repeated_reading_trial": "True when the paragraph is read for the second time.",
172 "is_correct": "Whether the comprehension question was answered correctly.",
173 "selected_answer": "The answer the participant chose.",
174 "question": "The comprehension question of the trial.",
175 "article_title": "The title of the article the paragraph comes from.",
176 "TRIAL_INDEX": "The trial's position in the participant's session.",
177 "is_in_aspan": "True for the words that answer the trial's question "
178 "(the answer span).",
179 "is_in_dspan": "True for the words of the distractor span.",
180}
182#: A short name for a highlight column in the figure's key (#374 F6).
183HIGHLIGHT_NOTES: dict[str, str] = {
184 "is_in_aspan": "answer span",
185 "is_in_dspan": "distractor span",
186}
189#: Canonical columns that can carry a copy of a partner's values — see
190#: `ColumnNames.aliases`.
191_ALIAS_PAIRS = (("trial_id", "unique_trial_id"), ("text_id", "unique_text_id"))
193#: The note a time column read in another unit carries ("FPOGD, in ms"). A
194#: figure drops it: its hover writes the unit after the value.
195IN_MS = ", in ms"
197#: The id columns a derived table (a summary, a saccade table) shares with the
198#: dataset's own — the only ones it may carry under the file's names.
199IDENTITY_COLUMNS = frozenset(
200 {
201 "participant_id",
202 "trial_id",
203 "unique_trial_id",
204 "text_id",
205 "unique_text_id",
206 "screen_id",
207 "word_id",
208 }
209)
212def canonical_label(column) -> str:
213 """A readable label for a column the app made (a measure, a run, an angle …)."""
214 from .data import READING_MEASURE_FIELDS
216 column = str(column)
217 for _key, canonical, _short, full, *_ in READING_MEASURE_FIELDS:
218 if canonical == column:
219 return str(full)
220 if column in _CANONICAL_LABELS:
221 return _CANONICAL_LABELS[column]
222 text = column.replace("_", " ").strip()
223 return text[:1].upper() + text[1:]
226@dataclass(frozen=True)
227class SourceName:
228 """What one canonical column was read from: its source column(s) and how."""
230 sources: tuple[str, ...]
231 kind: str = MAPPED
232 note: str = ""
234 @property
235 def display(self) -> str:
236 return " + ".join(self.sources)
239@dataclass(frozen=True)
240class ColumnNames:
241 """One table's canonical column → :class:`SourceName` record."""
243 entries: Mapping[str, SourceName] = field(default_factory=dict)
245 def source(self, column) -> SourceName | None:
246 return self.entries.get(str(column))
248 def kind_of(self, column) -> str:
249 entry = self.source(column)
250 if entry is not None:
251 return entry.kind
252 return COMPUTED if str(column) in COMPUTED_COLUMNS else YOURS
254 def display(self, column) -> str:
255 """The name to show for ``column`` — the user's when there is one."""
256 entry = self.source(column)
257 return entry.display if entry is not None and entry.sources else str(column)
259 def to_canonical(self, name) -> str:
260 """The canonical column a user's ``name`` stands for (else ``name``)."""
261 name = str(name)
262 for column, entry in self.entries.items():
263 if entry.sources == (name,) and entry.kind in (MAPPED, CONVERTED):
264 return column
265 return name
267 def through(self, earlier: ColumnNames) -> ColumnNames:
268 """This map with each source renamed by ``earlier``.
270 ✏️ Edit dataset maps fields onto the stored frame's *canonical* columns,
271 so a map built from that edit names canonical columns as its sources;
272 read through the dataset's earlier map they are the user's names again.
273 A source that was generated or converted stays so. Columns this map does
274 not rebuild keep their earlier record.
275 """
276 out: dict[str, SourceName] = {}
277 for column, entry in self.entries.items():
278 sources: list[str] = []
279 kind, note = entry.kind, entry.note
280 for source in entry.sources:
281 prior = earlier.entries.get(source)
282 if prior is None:
283 sources.append(source)
284 continue
285 sources.extend(prior.sources)
286 if kind == MAPPED and prior.kind != MAPPED:
287 kind, note = prior.kind, prior.note
288 out[column] = SourceName(tuple(sources), kind, note)
289 for column, entry in earlier.entries.items():
290 out.setdefault(column, entry)
291 return ColumnNames(out)
293 def label(self, column) -> str:
294 """What a person is shown for ``column`` (DATA-66 phase 2).
296 The user's own name when the column was read from their files; a
297 curated label marked :data:`COMPUTED_SUFFIX` when the app made it; else
298 the column's own name.
299 """
300 entry = self.source(column)
301 kind = self.kind_of(column)
302 if (
303 entry is not None
304 and entry.sources
305 and kind in (MAPPED, COMPOSITE, CONVERTED)
306 ):
307 # A converted column says what it holds now: a box width is the
308 # difference of two edges, not their sum, and a duration read in
309 # seconds is in ms.
310 if kind == CONVERTED and entry.note:
311 return entry.note
312 return entry.display
313 if kind == GENERATED and str(column) == "timestamp_ms":
314 # #374: no onset in the data — the stand-in is the order, not ms.
315 return "Fixation order" + COMPUTED_SUFFIX
316 if kind in (COMPUTED, GENERATED):
317 return canonical_label(column) + COMPUTED_SUFFIX
318 return str(column)
320 def field_label(self, column) -> str:
321 """What ``column`` is called where the app names a *field* — a chip, a
322 trial filter, a picker on the rail (#374 F5).
324 A role the dataset mapped (:data:`ROLE_LABELS`) is named by its role,
325 not by the source column; everything else as :meth:`label`.
326 """
327 column = str(column)
328 if column in ROLE_LABELS and self.kind_of(column) not in (
329 COMPUTED,
330 GENERATED,
331 ):
332 return ROLE_LABELS[column]
333 return self.label(column)
335 def source_tooltip(self, column) -> str:
336 """``"from RECORDING_SESSION_LABEL"`` for a role :meth:`field_label`
337 names by its role, else ``""`` (#374 F5)."""
338 column = str(column)
339 entry = self.source(column)
340 if column not in ROLE_LABELS or entry is None or not entry.sources:
341 return ""
342 if entry.kind == CONVERTED and entry.note:
343 return f"from {entry.note}"
344 if entry.kind not in (MAPPED, COMPOSITE):
345 return ""
346 return f"from {entry.display}"
348 def figure_labels(self, columns: Iterable) -> dict[str, str]:
349 """``{column: label}`` for a figure's text (`FigureSettings.column_labels`).
351 Every column the dataset brought, under its own name, except a mapped
352 role (:data:`ROLE_LABELS`); those and the columns the app made are left
353 out, so a figure keeps its short labels for them
354 ("Fixation #", "FFD") rather than writing "(computed)" into a hover.
355 Bookkeeping columns (`data.INTERNAL_COLUMNS`) are no figure's text. A
356 column converted from one source (a duration read in seconds) is named
357 by that source: the figure writes its unit after the value, and the
358 note's "…, in ms" would say it twice.
359 """
360 from .data import INTERNAL_COLUMNS
362 out: dict[str, str] = {}
363 for column in columns:
364 # #374 F5/F7: a role is the figure's own plain word ("Duration",
365 # "Word #"), not the source column.
366 if (
367 column in INTERNAL_COLUMNS
368 or str(column) in ROLE_LABELS
369 or self.kind_of(column) in (COMPUTED, GENERATED)
370 ):
371 continue
372 entry = self.source(column)
373 unit_conversion = (
374 entry is not None
375 and entry.kind == CONVERTED
376 and entry.note.endswith(IN_MS)
377 )
378 out[str(column)] = (
379 entry.note.removesuffix(IN_MS)
380 if unit_conversion
381 else self.label(column)
382 )
383 return out
385 def merged(self, other: ColumnNames) -> ColumnNames:
386 """Both tables' entries, this map's winning where both name a column."""
387 return ColumnNames({**dict(other.entries), **dict(self.entries)})
389 def option_labels(
390 self, options, extra: Mapping | None = None, *, roles: bool = False
391 ) -> dict[str, str]:
392 """``{option: label}`` for a picker; ``extra`` labels synthetic options
393 (``"(uniform)"``, ``"line"``). A label two options share gets the
394 internal name added, so a picker never shows two identical rows.
395 ``roles=True`` names a mapped role by its role (:meth:`field_label`) —
396 for the rail's pickers, not the mapping screens."""
397 extra = dict(extra or {})
398 name = self.field_label if roles else self.label
399 labels = {o: extra.get(o) or name(o) for o in options}
400 counts: dict[str, int] = {}
401 for value in labels.values():
402 counts[value] = counts.get(value, 0) + 1
403 return {
404 o: f"{label} · {o}" if counts[label] > 1 and label != str(o) else label
405 for o, label in labels.items()
406 }
408 def sort_options(self, options, first=()) -> list:
409 """The user's columns first, the app's last; ``first`` stays in front.
411 Stable within each group, so a curated order survives."""
412 options = list(options)
413 head = [o for o in options if o in first]
414 rest = [o for o in options if o not in first]
415 made = (COMPUTED, GENERATED)
416 return (
417 head
418 + [o for o in rest if self.kind_of(o) not in made]
419 + [o for o in rest if self.kind_of(o) in made]
420 )
422 def aliases(self, columns: Iterable[str]) -> set[str]:
423 """Columns in ``columns`` that only repeat a partner from the same source.
425 `unique_trial_id` mirrors `trial_id` (BUG-58) and `unique_text_id`
426 usually mirrors `text_id`; when both of a pair were read from one column
427 of the user's file, a table needs to show it once.
428 """
429 present = {str(c) for c in columns}
430 hidden: set[str] = set()
431 for main, alias in _ALIAS_PAIRS:
432 first, second = self.source(main), self.source(alias)
433 if (
434 main in present
435 and alias in present
436 and first is not None
437 and second is not None
438 and first.sources
439 and first.sources == second.sources
440 ):
441 hidden.add(alias)
442 return hidden
444 def with_rewrites(self, table: str, rewrites: Iterable) -> ColumnNames:
445 """This map with every column the load rewrote (``data.Rewrite``s for
446 ``table``) marked converted: a padded id, a shifted word id, positions
447 filled from word boxes are no longer what the file held under its name,
448 so the column is labelled "<source><how>" and exported under its
449 internal name."""
450 changed = dict(self.entries)
451 for rewritten_table, column, how in rewrites or ():
452 entry = changed.get(column)
453 if rewritten_table != table or entry is None or entry.kind != MAPPED:
454 continue
455 note = " + ".join(entry.sources) + how
456 changed[column] = SourceName(entry.sources, CONVERTED, note)
457 return ColumnNames(changed)
459 def identity(self) -> ColumnNames:
460 """This map's id columns only (:data:`IDENTITY_COLUMNS`).
462 For a table the app derives — a summary, a saccade table, a character
463 grid — whose other columns reuse a canonical name (`n_fixations`,
464 `duration_ms`, `x`) for a value of their own: only its ids are the
465 file's."""
466 return self.restricted_to(IDENTITY_COLUMNS)
468 def redundant_aliases(self, frame) -> set[str]:
469 """:meth:`aliases` that really repeat their partner in ``frame``.
471 The map says both came from one column; the values decide, since a
472 padded or suffixed id can part from its copy (BUG-59)."""
473 partners = {alias: main for main, alias in _ALIAS_PAIRS}
474 return {
475 alias
476 for alias in self.aliases(frame.columns)
477 if frame[alias].equals(frame[partners[alias]])
478 }
480 def export_headers(self, columns: Iterable) -> dict[str, str]:
481 """``{column: header}`` for a table written out (DATA-66 phase 3).
483 A column read from one column of the user's file is written under that
484 column's name. One built from several (a composite trial id), converted
485 (a box width from two edges, a duration read in seconds) or made by the
486 app keeps its canonical name — its values are not what the file held
487 under any one name — and the bundle's `columns.json` and README say
488 where it came from. A header that would repeat another column's name
489 keeps the canonical one. Columns not renamed are left out.
490 """
491 names = [str(c) for c in columns]
492 taken = set(names)
493 out: dict[str, str] = {}
494 for column in names:
495 entry = self.source(column)
496 if entry is None or entry.kind != MAPPED or len(entry.sources) != 1:
497 continue
498 header = entry.sources[0]
499 if header == column or header in taken:
500 continue
501 taken.add(header)
502 out[column] = header
503 return out
505 def provenance(self, column) -> str:
506 """Where ``column`` came from, in a sentence fragment (the README's)."""
507 entry = self.source(column)
508 kind = self.kind_of(column)
509 sources = ", ".join(f"`{s}`" for s in entry.sources) if entry else ""
510 note = entry.note if entry else ""
511 if kind == MAPPED and sources:
512 return f"your column {sources}"
513 if kind == COMPOSITE:
514 return f"joined from your columns {sources}"
515 if kind == CONVERTED:
516 return f"converted from your {sources}" + (f" ({note})" if note else "")
517 if kind == GENERATED:
518 return "made by Scanpath Studio" + (f" ({note})" if note else "")
519 if kind == COMPUTED:
520 return "computed by Scanpath Studio"
521 return "your column, under its own name"
523 def restricted_to(self, columns: Iterable[str]) -> ColumnNames:
524 """Only the entries for ``columns`` — a frame's actual columns.
526 After an edit, :meth:`through` keeps every earlier entry the edit did
527 not rebuild, including one for a column the edit removed (a cleared
528 reading measure); restricting to the saved frame drops those, so the
529 record never names a column the dataset no longer has.
530 """
531 keep = {str(c) for c in columns}
532 return ColumnNames({c: e for c, e in self.entries.items() if c in keep})
534 def to_payload(self) -> dict:
535 """A JSON-safe form, for a `_datasets` entry and the recovery cache."""
536 return {
537 column: {"sources": list(e.sources), "kind": e.kind, "note": e.note}
538 for column, e in self.entries.items()
539 }
541 @classmethod
542 def from_payload(cls, payload) -> ColumnNames:
543 """The inverse of :meth:`to_payload`; anything malformed reads as empty."""
544 if not isinstance(payload, Mapping):
545 return EMPTY
546 entries: dict[str, SourceName] = {}
547 for column, raw in payload.items():
548 if not isinstance(raw, Mapping):
549 return EMPTY
550 kind = raw.get("kind", MAPPED)
551 sources = raw.get("sources") or ()
552 if kind not in KINDS or not isinstance(sources, Iterable):
553 return EMPTY
554 entries[str(column)] = SourceName(
555 tuple(str(s) for s in sources), kind, str(raw.get("note") or "")
556 )
557 return cls(entries)
560EMPTY = ColumnNames({})
563def active(session: Mapping, table: str) -> ColumnNames:
564 """The open dataset's map for ``table``, read from a session mapping.
566 Takes the session as an argument so this module stays free of Streamlit;
567 callers pass ``st.session_state``.
568 """
569 stash = session.get(ACTIVE_COLUMN_NAMES_KEY) or {}
570 return ColumnNames.from_payload(stash.get(table))
573def active_all(session: Mapping) -> ColumnNames:
574 """The open dataset's map over all its tables, for a label any table can own.
576 The fixations table's entries win, then the words table's (word-level
577 fields such as surprisal are carried onto fixations), then raw gaze's.
578 """
579 return across_tables({table: active(session, table) for table in _TABLES})
582def table_figure_labels(
583 maps: Mapping[str, ColumnNames], columns: Mapping[str, Iterable]
584) -> dict[str, str]:
585 """`FigureSettings.column_labels` for a figure drawn from several tables.
587 One flat map cannot hold two names for one column — ``word_id`` is the AOI
588 table's ``IA_ID`` and the fixation table's interest-area column — so the
589 merged map's labels (fixations' winning) sit under the plain keys and a
590 word column the words table names differently also under
591 ``"words:<column>"``, which a word hover reads first (`plots._table_label`).
592 Only the tables the figure is drawn from are consulted, so a words-only
593 figure is labelled by the words table alone.
594 """
595 columns = {table: list(cols) for table, cols in columns.items()}
596 drawn = {table: names for table, names in maps.items() if table in columns}
597 out = across_tables(drawn).figure_labels(
598 [column for cols in columns.values() for column in cols]
599 )
600 words = maps.get("words")
601 if words is not None and "words" in columns:
602 for column, label in words.figure_labels(columns["words"]).items():
603 if out.get(column) != label:
604 out[f"words:{column}"] = label
605 return out
608def active_figure_labels(session: Mapping, **columns: Iterable) -> dict[str, str]:
609 """:func:`table_figure_labels` for the open dataset — ``columns`` keyed by
610 table (``words=…``, ``fixations=…``)."""
611 return table_figure_labels(
612 {table: active(session, table) for table in _TABLES}, columns
613 )
616#: The tables a dataset's map covers, in the order their entries win.
617_TABLES = ("fixations", "words", "raw_gaze")
620def across_tables(maps: Mapping[str, ColumnNames]) -> ColumnNames:
621 """One map from a dataset's per-table maps (fixations', then words', then
622 raw gaze's entries win) — what :func:`active_all` reads from the session,
623 for a dataset held elsewhere (Compare's B, `SecondaryDataset.column_names`)."""
624 out = EMPTY
625 for table in _TABLES:
626 if table in maps:
627 out = out.merged(maps[table])
628 return out
631#: Mapped screen fields: schema key → canonical column (`data._copy_screen_fields`).
632_SCREEN_FIELDS = (
633 ("screen_id", "screen_id"),
634 ("screen_index", "screen_index"),
635 ("screen_timestamp", "screen_timestamp_ms"),
636 ("screen_fixation_id", "screen_fixation_id"),
637 ("canvas_width", "canvas_width"),
638 ("canvas_height", "canvas_height"),
639)
642def _id_entry(value) -> SourceName | None:
643 """A schema id field — one column, or several joined (a composite id)."""
644 from .data import trial_mapping_columns
646 if not value:
647 return None
648 columns = [str(c) for c in trial_mapping_columns(value)]
649 if len(columns) > 1:
650 return SourceName(tuple(columns), COMPOSITE)
651 return SourceName((columns[0],), MAPPED)
654def _timed(column: str) -> SourceName:
655 """A time column, which normalization converts to ms when its unit says so."""
656 from .data import time_unit_ms
658 if time_unit_ms(column) != 1.0:
659 return SourceName((column,), CONVERTED, f"{column}, in ms")
660 return SourceName((column,), MAPPED)
663def _mapped_or(schema: Mapping, key: str, kind: str, note: str) -> SourceName:
664 """The schema's column for ``key``, else a stand-in of ``kind``."""
665 column = schema.get(key)
666 return SourceName((str(column),)) if column else SourceName((), kind, note)
669def _registry(out: dict, registry, present: set, keep: set | None) -> None:
670 """The optional-field renames normalization applied (`_apply_optional_fields`).
672 A later entry for the same destination overwrites an earlier one there, so
673 it does here too; a passthrough (`src == dest`) keeps its own name and needs
674 no entry.
675 """
676 for src, dest, _kind, _category in registry:
677 if src not in present or (keep is not None and src not in keep):
678 continue
679 if dest != src:
680 out[dest] = SourceName((src,), MAPPED)
683def from_schema(
684 table: str,
685 schema: Mapping | None,
686 columns: Iterable[str],
687 *,
688 keep_columns: Iterable[str] | None = None,
689) -> ColumnNames:
690 """The map ``data.normalize_<table>`` implies for ``schema`` over ``columns``.
692 ``table`` is ``"words"``, ``"fixations"`` or ``"raw_gaze"``; ``columns`` are
693 the raw table's; ``keep_columns`` the set normalization was given (``None``
694 carries every registry field, as normalization does).
695 """
696 from . import data
698 schema = dict(schema or {})
699 present = {str(c) for c in columns}
700 keep = None if keep_columns is None else {str(c) for c in keep_columns}
701 out: dict[str, SourceName] = {}
703 out["participant_id"] = _id_entry(schema.get("participant")) or SourceName(
704 (), GENERATED, "one participant for the whole table"
705 )
706 if trial := _id_entry(schema.get("trial")):
707 out["trial_id"] = out["unique_trial_id"] = trial
708 if table != "raw_gaze" and "unique_paragraph_id" in present:
709 out["text_id"] = out["unique_text_id"] = SourceName(("unique_paragraph_id",))
710 elif text_id := _id_entry(schema.get("text_id")):
711 # A remap fills `unique_text_id` from the mapped Text ID
712 # (`data.remap_normalized_frame`); a first load has no such column, and
713 # an entry for an absent column names nothing. A `unique_text_id` the
714 # file itself has is the user's own column, under its own name.
715 out["text_id"] = text_id
716 if "unique_text_id" not in present:
717 out["unique_text_id"] = text_id
718 else:
719 out["text_id"] = SourceName((), GENERATED, "the trial id")
720 for key, canonical in _SCREEN_FIELDS:
721 if schema.get(key):
722 out[canonical] = SourceName((str(schema[key]),))
724 if table == "words":
725 out["word_id"] = _mapped_or(schema, "word_id", GENERATED, "the row order")
726 out["text"] = _mapped_or(schema, "text", GENERATED, "w0, w1 … from the word id")
727 out["line_idx"] = _mapped_or(schema, "line", GENERATED, "1 — one line")
728 if all(schema.get(k) for k in ("x", "y", "width", "height")):
729 for key in ("x", "y", "width", "height"):
730 out[key] = SourceName((str(schema[key]),))
731 elif all(schema.get(k) for k in ("left", "right", "top", "bottom")):
732 left, right = str(schema["left"]), str(schema["right"])
733 top, bottom = str(schema["top"]), str(schema["bottom"])
734 out["x"], out["y"] = SourceName((left,)), SourceName((top,))
735 out["width"] = SourceName((right, left), CONVERTED, f"{right} − {left}")
736 out["height"] = SourceName((bottom, top), CONVERTED, f"{bottom} − {top}")
737 _registry(out, data.WORD_OPTIONAL_FIELDS, present, keep)
738 # AN-32: a measure the schema names decides it, after the passthrough.
739 for key, canonical, *_ in data.READING_MEASURE_FIELDS:
740 if key not in schema:
741 continue
742 column = schema.get(key)
743 if column and column in present:
744 out[canonical] = SourceName((str(column),))
745 else:
746 out.pop(canonical, None)
747 elif table == "fixations":
748 for coord in ("x", "y"):
749 out[coord] = _mapped_or(
750 schema, coord, COMPUTED, "the fixated word's box center"
751 )
752 if schema.get("duration"):
753 out["duration_ms"] = _timed(str(schema["duration"]))
754 out["timestamp_ms"] = (
755 _timed(str(schema["timestamp"]))
756 if schema.get("timestamp")
757 else SourceName((), GENERATED, "the fixation's order in its trial")
758 )
759 out["fixation_id"] = _mapped_or(
760 schema, "fixation_id", GENERATED, "1, 2, … per trial"
761 )
762 out["word_id"] = _mapped_or(
763 schema, "word_id", COMPUTED, "assigned from the word boxes"
764 )
765 _registry(out, data.FIX_OPTIONAL_FIELDS, present, keep)
766 else: # raw gaze
767 for key in ("x", "y", "text", "word_id"):
768 if schema.get(key):
769 out[key] = SourceName((str(schema[key]),))
770 if schema.get("timestamp"):
771 out["timestamp_ms"] = _timed(str(schema["timestamp"]))
772 else:
773 # No clock: the samples are numbered, never given a time.
774 out[SAMPLE_INDEX] = SourceName(
775 (), GENERATED, "the sample's order in its trial (no time mapped)"
776 )
777 return ColumnNames(out)
780def for_tables(
781 schemas: Mapping[str, Mapping | None],
782 frames: Mapping[str, object],
783 keeps: Mapping[str, Iterable[str] | None] | None = None,
784 rewrites: Iterable | None = None,
785) -> dict[str, dict]:
786 """``{table: payload}`` for every table with a schema and a raw frame.
788 ``rewrites`` are the ``data.Rewrite``s the load made
789 (:meth:`ColumnNames.with_rewrites`)."""
790 keeps = keeps or {}
791 rewrites = tuple(rewrites or ())
792 out: dict[str, dict] = {}
793 for table, schema in schemas.items():
794 columns = getattr(frames.get(table), "columns", None)
795 if not schema or columns is None or len(columns) == 0:
796 continue
797 names = from_schema(table, schema, columns, keep_columns=keeps.get(table))
798 out[table] = names.with_rewrites(table, rewrites).to_payload()
799 return out
802# --- Phase 3: what a written table calls its columns -------------------------
804#: The `columns.json` format a bundle writes beside its tables.
805COLUMNS_FILE_SCHEMA = 1
808def _written_plan(frame, names: ColumnNames) -> tuple[set[str], dict[str, str]]:
809 """``(left out, renamed)`` for writing ``frame``: the `unique_*` aliases that
810 only repeat their partner, and each remaining column's header."""
811 hidden = names.redundant_aliases(frame)
812 kept = [c for c in frame.columns if c not in hidden]
813 return hidden, names.export_headers(kept)
816def as_written(frame, names: ColumnNames | None, hidden: set[str] | None = None):
817 """``frame`` as a bundle writes it: each column the user's file named under
818 that name (`ColumnNames.export_headers`), and a `unique_*` alias that only
819 repeats its partner left out. The same object when nothing changes.
821 ``hidden`` is the aliases to leave out, decided once for the whole table
822 (:meth:`ColumnNames.redundant_aliases`) so every per-trial file of it has
823 the same columns; without it, ``frame`` decides."""
824 if names is None or not names.entries:
825 return frame
826 if hidden is None:
827 hidden, headers = _written_plan(frame, names)
828 else:
829 hidden = {c for c in hidden if c in frame.columns}
830 headers = names.export_headers([c for c in frame.columns if c not in hidden])
831 if hidden:
832 frame = frame.drop(columns=sorted(hidden))
833 return frame.rename(columns=headers) if headers else frame
836def written_columns(frame, names: ColumnNames | None) -> list[dict]:
837 """What :func:`as_written` makes of ``frame``'s mapped columns, one row each:
838 the header written, the internal (canonical) name, its kind, its sources and
839 note. Columns the map does not record are written under their own names and
840 are not listed."""
841 if names is None or not names.entries:
842 return []
843 hidden, headers = _written_plan(frame, names)
844 rows = []
845 for column in frame.columns:
846 entry = names.source(column)
847 if column in hidden or entry is None:
848 continue
849 rows.append(
850 {
851 "column": headers.get(column, column),
852 "canonical": column,
853 "kind": entry.kind,
854 "sources": list(entry.sources),
855 "note": entry.note,
856 }
857 )
858 return rows
861def columns_manifest(tables: Mapping[str, list[dict]]) -> dict:
862 """The `columns.json` a bundle carries: ``{table: written_columns(…)}``, so a
863 script can map every file back to the internal names."""
864 return {
865 "schema": COLUMNS_FILE_SCHEMA,
866 "tables": {table: rows for table, rows in tables.items() if rows},
867 }
870def dictionary_lines(
871 tables: Mapping[str, list[dict]], names: Mapping[str, ColumnNames]
872) -> list[str]:
873 """The README's data dictionary: for each table's :func:`written_columns`,
874 the header written and where it came from, in markdown."""
875 titles = {
876 "fixations": "Fixations",
877 "words": "Words (interest areas)",
878 "raw_gaze": "Raw gaze",
879 }
880 lines: list[str] = []
881 for table, rows in tables.items():
882 if not rows:
883 continue
884 lines += ["", f"### {titles.get(table, table)}"]
885 for row in rows:
886 written, column = row["column"], row["canonical"]
887 internal = "" if written == column else f" (internally `{column}`)"
888 lines.append(f"- `{written}`{internal}: {names[table].provenance(column)}")
889 return lines
892def source_schema(
893 schema: Mapping | None, names: ColumnNames
894) -> tuple[dict | None, tuple[str, ...]]:
895 """``schema`` restated in the dataset's own files' column names.
897 ✏️ Edit dataset maps fields onto the stored frame's *canonical* columns
898 (``{"trial": "trial_id", "x": "x"}``), which say nothing to a script that
899 reads the original files. Read through ``names`` — the map from before
900 that edit — each becomes the column(s) it was read from, so the result can
901 be handed to ``api.load_scanpath_data`` over those files (Share → Code). A
902 column the app *made* (a generated stand-in, a computed value) maps to
903 nothing, which makes the loader make it again; a box stored as
904 ``x/y/width/height`` but read from edges is restated as those edges.
906 Returns ``(schema, unresolved)``: the fields that could not be traced back,
907 left as they were, for the caller to name.
908 """
909 from .data import trial_mapping_columns
911 if schema is None:
912 return None, ()
913 out: dict = {}
914 unresolved: list[str] = []
915 for key, value in schema.items():
916 if not value:
917 out[key] = value
918 continue
919 columns: list[str] = []
920 made = traced = False
921 for column in trial_mapping_columns(value):
922 entry = names.source(column)
923 if entry is None:
924 columns.append(str(column))
925 elif entry.kind in (GENERATED, COMPUTED) or not entry.sources:
926 made = True
927 elif entry.kind == CONVERTED and len(entry.sources) != 1:
928 traced = True # right − left: only the edges say it
929 else:
930 columns.extend(entry.sources)
931 if traced:
932 out[key] = value
933 unresolved.append(str(key))
934 elif made:
935 out[key] = None
936 elif isinstance(value, str) and len(columns) == 1:
937 out[key] = columns[0]
938 else:
939 out[key] = columns
940 # A box read from edges is stored as x/y/width/height, its width and height
941 # computed as right − left: restate it as the edges it came from.
942 sizes = [names.source(schema.get(side) or "") for side in ("width", "height")]
943 if (
944 {"width", "height"} <= set(unresolved)
945 and all(e is not None and len(e.sources) == 2 for e in sizes)
946 and isinstance(out.get("x"), str)
947 and isinstance(out.get("y"), str)
948 ):
949 out.update(
950 left=out["x"],
951 top=out["y"],
952 right=sizes[0].sources[0],
953 bottom=sizes[1].sources[0],
954 x=None,
955 y=None,
956 width=None,
957 height=None,
958 )
959 unresolved = [k for k in unresolved if k not in ("width", "height")]
960 return out, tuple(unresolved)
963def stored_source_recipe(stored: Mapping) -> dict:
964 """A stored dataset's ``source_recipe`` — how a script loads its files.
966 Written when the dataset is added (`wizard._source_recipe`) and kept
967 current by ✏️ Edit dataset; read by Share → Code
968 (`code_snippet.upload_source`). A dataset stored before the recipe existed
969 gets one read off its stored mapping through its column-name map, which is
970 right unless one of its files used a canonical column name for a different
971 field.
972 """
973 recipe = stored.get("source_recipe")
974 if isinstance(recipe, Mapping):
975 return dict(recipe)
976 names = stored.get("column_names") or {}
977 schemas: dict = {}
978 unresolved: dict = {}
979 for table, schema in (stored.get("schemas") or {}).items():
980 schemas[table], missing = source_schema(
981 schema, ColumnNames.from_payload(names.get(table))
982 )
983 if missing:
984 unresolved[table] = list(missing)
985 return {"schemas": schemas, "unresolved": unresolved}
988# --- Phase 4: a frame that carries its own names ------------------------------
990#: The `DataFrame.attrs` key a frame under the dataset's own names carries its
991#: map in (`attach`). pandas 3 keeps `attrs` through filtering, `loc`, `copy`,
992#: `assign` and `groupby`, so a script can slice the frame it loaded and hand it
993#: back to the API. `merge` and `concat` keep them only when every input carries
994#: the same `attrs`: a frame joined with a table of the user's own loses its map,
995#: and `api._require_normalized` says how to get it back.
996ATTRS_KEY = "scanpath_studio.columns"
999def attach(frame, table: str, names: ColumnNames | None):
1000 """``frame`` under the dataset's own names (:func:`as_written`), carrying
1001 its map in ``attrs`` so :func:`to_canonical_frame` can undo it. A frame
1002 with no names to apply comes back as it was."""
1003 if frame is None or names is None or not names.entries:
1004 return frame
1005 hidden, headers = _written_plan(frame, names)
1006 partners = {alias: main for main, alias in _ALIAS_PAIRS}
1007 out = as_written(frame, names, hidden)
1008 # One JSON string, not a nested dict: pandas deep-copies `attrs` on every
1009 # operation, and walking a dict of dicts made a script's per-trial loop
1010 # over a corpus 2-3x slower; a string is copied for nothing.
1011 record = {
1012 "table": table,
1013 "names": names.to_payload(),
1014 "renamed": {header: column for column, header in headers.items()},
1015 "aliases": {alias: partners[alias] for alias in hidden},
1016 }
1017 out.attrs = {**frame.attrs, ATTRS_KEY: json.dumps(record)}
1018 return out
1021def _record(frame) -> dict | None:
1022 """The record :func:`attach` left on ``frame``, else ``None``."""
1023 raw = getattr(frame, "attrs", {}).get(ATTRS_KEY)
1024 if not isinstance(raw, str):
1025 return None
1026 try:
1027 record = json.loads(raw)
1028 except ValueError:
1029 return None
1030 return record if isinstance(record, dict) else None
1033def frame_names(frame) -> tuple[str, ColumnNames] | None:
1034 """``(table, map)`` a frame under the dataset's own names carries, else
1035 ``None`` (a canonical frame, or any other table)."""
1036 record = _record(frame)
1037 if record is None:
1038 return None
1039 return str(record.get("table", "")), ColumnNames.from_payload(record.get("names"))
1042def to_canonical_frame(frame):
1043 """A frame :func:`attach` named, back under the canonical names it is
1044 processed in — the inverse every API function applies on entry. Any other
1045 frame is returned unchanged."""
1046 record = _record(frame)
1047 if record is None:
1048 return frame
1049 renamed = {
1050 header: column
1051 for header, column in (record.get("renamed") or {}).items()
1052 if header in frame.columns and column not in frame.columns
1053 }
1054 out = frame.rename(columns=renamed) if renamed else frame.copy(deep=False)
1055 for alias, main in (record.get("aliases") or {}).items():
1056 if alias not in out.columns and main in out.columns:
1057 out[alias] = out[main]
1058 out.attrs = {k: v for k, v in frame.attrs.items() if k != ATTRS_KEY}
1059 return out