Coverage for scanpath_studio/multipart.py: 98%
182 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Canonical identity and validation for trials made of ordered screens.
3Legacy Scanpath Studio data identifies one coordinate space with
4``(participant_id, trial_id)``. Multipart trials retain that logical parent
5identity and add ``screen_id`` plus a 1-based ``screen_index`` on every row.
6This module is deliberately pandas-only so ingestion, measures, UI, API, CLI,
7and export can share the exact same boundary rules without importing Streamlit.
8"""
10from __future__ import annotations
12from collections.abc import Mapping, Sequence
13from typing import Any
15import numpy as np
16import pandas as pd
18PARENT_KEY = ("participant_id", "trial_id")
19SCREEN_ID = "screen_id"
20SCREEN_INDEX = "screen_index"
21PART_KEY = (*PARENT_KEY, SCREEN_ID)
22SCREEN_TIMESTAMP = "screen_timestamp_ms"
23SCREEN_FIXATION_ID = "screen_fixation_id"
24CANVAS_WIDTH = "canvas_width"
25CANVAS_HEIGHT = "canvas_height"
28def has_screen_identity(frame: pd.DataFrame | None) -> bool:
29 """Whether ``frame`` carries canonical screen identity."""
30 return bool(frame is not None and SCREEN_ID in frame.columns)
33def grouping_columns(frame: pd.DataFrame, *, include_word: bool = False) -> list[str]:
34 """Identity columns for a scientific operation on ``frame``.
36 Single-screen frames keep the historical parent key. Multipart frames add
37 ``screen_id`` so assignment, saccades, runs, passes, and measures can never
38 connect two coordinate spaces accidentally.
39 """
40 keys = [column for column in PARENT_KEY if column in frame.columns]
41 if has_screen_identity(frame):
42 keys.append(SCREEN_ID)
43 if include_word and "word_id" in frame.columns:
44 keys.append("word_id")
45 return keys
48def _parent_columns(frame: pd.DataFrame) -> list[str]:
49 missing = [column for column in PARENT_KEY if column not in frame.columns]
50 if missing:
51 raise ValueError(
52 "Multipart identity needs canonical parent columns: " + ", ".join(missing)
53 )
54 return list(PARENT_KEY)
57def normalize_screen_identity(frame: pd.DataFrame) -> pd.DataFrame:
58 """Normalize/validate screen columns, preserving legacy frame shape.
60 ``screen_id`` alone is enough: order is derived from first appearance
61 within each logical trial. ``screen_index`` alone is also accepted and its
62 string value becomes the id. When both are supplied, the mapping must be
63 one-to-one inside a parent. Canvas dimensions, when supplied, must be
64 positive and constant within a screen.
65 """
66 has_id = SCREEN_ID in frame.columns
67 has_index = SCREEN_INDEX in frame.columns
68 if not has_id and not has_index:
69 return frame
71 parents = _parent_columns(frame)
72 out = frame.copy()
73 if not has_id:
74 numeric_index = pd.to_numeric(out[SCREEN_INDEX], errors="coerce")
75 if numeric_index.isna().any():
76 raise ValueError(
77 "`screen_index` (screen order) has blank or non-numeric cells."
78 )
79 out[SCREEN_ID] = numeric_index.astype(int).astype(str)
80 else:
81 # `stable_id`, not a plain `.astype(str)` — BUG-44's hazard applies here
82 # too: a whole-number screen_id reads as float64 the moment any OTHER
83 # row anywhere in that column is missing, so one report's `"1"` becomes
84 # another's `"1.0"` and `validate_matching_parts` below rejects every
85 # screen as an orphan even though both sides recorded the same one.
86 # Imported locally — `data.py` imports from this module, so a
87 # module-level import would cycle.
88 from .data import stable_id
90 # A missing/blank cell is tolerated exactly as `trial_id_series` (via
91 # `stable_id`) already tolerates one in a trial id — no proactive
92 # `isna()` check here either. Rejecting it outright meant a screen_id
93 # column with a stray blank cell (the same dtype-coercion quirk
94 # BUG-44 fixed for identity columns generally) failed the whole
95 # mapping instead of just reading that one blank cell as "nan".
96 out[SCREEN_ID] = stable_id(out[SCREEN_ID])
98 if not has_index:
99 distinct = out[parents + [SCREEN_ID]].drop_duplicates()
100 distinct[SCREEN_INDEX] = distinct.groupby(parents, sort=False).cumcount() + 1
101 out = out.merge(distinct, on=parents + [SCREEN_ID], how="left", sort=False)
102 else:
103 numeric_index = pd.to_numeric(out[SCREEN_INDEX], errors="coerce")
104 if numeric_index.isna().any() or (numeric_index <= 0).any():
105 raise ValueError("`screen_index` (screen order) must count from 1.")
106 if (numeric_index % 1 != 0).any():
107 raise ValueError("`screen_index` (screen order) must be whole numbers.")
108 out[SCREEN_INDEX] = numeric_index.astype(int)
110 pairs = out[parents + [SCREEN_ID, SCREEN_INDEX]].drop_duplicates()
111 if pairs.duplicated(parents + [SCREEN_ID], keep=False).any():
112 raise ValueError("Within a trial, one Screen ID has two `screen_index` values.")
113 if pairs.duplicated(parents + [SCREEN_INDEX], keep=False).any():
114 raise ValueError("Within a trial, two Screen IDs share one `screen_index`.")
116 for column in (CANVAS_WIDTH, CANVAS_HEIGHT):
117 if column not in out.columns:
118 continue
119 values = pd.to_numeric(out[column], errors="coerce")
120 if values.notna().any() and (values.dropna() <= 0).any():
121 raise ValueError(f"{_FIELD_LABELS[column]} must be positive.")
122 out[column] = values
123 counts = out.groupby(list(PART_KEY), dropna=False)[column].nunique(dropna=True)
124 if (counts > 1).any():
125 raise ValueError(f"{_FIELD_LABELS[column]} changes within one screen.")
126 return out
129#: What a user calls each screen column in a message (the add screen's names).
130_FIELD_LABELS = {
131 SCREEN_INDEX: "`screen_index`",
132 CANVAS_WIDTH: "Screen canvas width",
133 CANVAS_HEIGHT: "Screen canvas height",
134}
137def _screens(parts: list) -> str:
138 """Up to three ``(participant, trial, screen)`` keys, as ``p01/3/page_2``."""
139 return ", ".join("/".join(str(v) for v in part) for part in parts[:3])
142def part_catalog(*frames: pd.DataFrame | None) -> pd.DataFrame:
143 """One ordered row per screen across ``frames``.
145 Metadata conflicts between words and fixations are rejected rather than
146 resolved with a silent ``first()``. Legacy data returns an empty catalogue.
147 """
148 rows: list[pd.DataFrame] = []
149 metadata = [SCREEN_INDEX, CANVAS_WIDTH, CANVAS_HEIGHT, "text_id"]
150 for frame in frames:
151 if frame is None or frame.empty or not has_screen_identity(frame):
152 continue
153 normalized = normalize_screen_identity(frame)
154 columns = [*PART_KEY, *(c for c in metadata if c in normalized.columns)]
155 candidate = normalized[columns].drop_duplicates()
156 for column in metadata:
157 if column not in candidate.columns:
158 candidate[column] = pd.NA
159 rows.append(candidate[[*PART_KEY, *metadata]])
160 if not rows:
161 return pd.DataFrame(columns=[*PART_KEY, *metadata])
163 combined = pd.concat(rows, ignore_index=True)
164 for column in metadata:
165 conflicts = combined.groupby(list(PART_KEY), dropna=False)[column].nunique(
166 dropna=True
167 )
168 if (conflicts > 1).any():
169 # tabs.py matches this text exactly (the screen_index case), so it
170 # keeps its wording until that check reads something sturdier (#374).
171 raise ValueError(f"Multipart metadata {column!r} conflicts across tables.")
172 catalog = (
173 combined.groupby(list(PART_KEY), as_index=False, dropna=False)
174 .first()
175 .sort_values([*PARENT_KEY, SCREEN_INDEX, SCREEN_ID], kind="stable")
176 .reset_index(drop=True)
177 )
178 return catalog
181def validate_matching_parts(words: pd.DataFrame, fixations: pd.DataFrame) -> None:
182 """Reject orphan screens when both normalized reports carry part identity."""
183 if words.empty or fixations.empty:
184 return
185 if not has_screen_identity(words) and not has_screen_identity(fixations):
186 return
187 if has_screen_identity(words) != has_screen_identity(fixations):
188 raise ValueError(
189 "Screen ID is set in only one table; set it in both or neither."
190 )
191 word_parts = set(map(tuple, words[list(PART_KEY)].drop_duplicates().to_numpy()))
192 fixation_parts = set(
193 map(tuple, fixations[list(PART_KEY)].drop_duplicates().to_numpy())
194 )
195 if word_parts != fixation_parts:
196 missing_words = sorted(fixation_parts - word_parts)
197 missing_fix = sorted(word_parts - fixation_parts)
198 details = []
199 if missing_words:
200 details.append(f"no words for {_screens(missing_words)}")
201 if missing_fix:
202 details.append(f"no fixations for {_screens(missing_fix)}")
203 raise ValueError("Some screens are in one table only: " + "; ".join(details))
206def extract_part(
207 frame: pd.DataFrame,
208 participant_id: Any,
209 trial_id: Any,
210 screen_id: Any | None = None,
211) -> pd.DataFrame:
212 """Extract one parent trial or one screen without concatenating screens."""
213 if frame is None or frame.empty:
214 return frame
215 mask = (frame["participant_id"].astype(str) == str(participant_id)) & (
216 frame["trial_id"].astype(str) == str(trial_id)
217 )
218 if screen_id is not None:
219 if SCREEN_ID not in frame.columns:
220 raise ValueError("screen= was supplied for a single-screen dataset.")
221 mask &= frame[SCREEN_ID].astype(str) == str(screen_id)
222 return frame.loc[mask]
225def _manifest_trials(manifest: Mapping[str, Any] | Sequence[Mapping[str, Any]]) -> list:
226 if isinstance(manifest, Mapping):
227 trials = manifest.get("trials", manifest.get("multipart_trials", []))
228 else:
229 trials = manifest
230 if not isinstance(trials, Sequence) or isinstance(trials, (str, bytes)):
231 raise ValueError("trial_parts_manifest must contain a 'trials' list.")
232 return list(trials)
235def apply_trial_parts_manifest(
236 normalized: pd.DataFrame,
237 source: pd.DataFrame,
238 manifest: Mapping[str, Any] | Sequence[Mapping[str, Any]],
239 *,
240 kind: str,
241) -> pd.DataFrame:
242 """Attach a nested trial-parts manifest to a normalized report.
244 Each part declares ``screen_id``, optional ``screen_index``/canvas size, and
245 a source-row selector under ``words`` or ``fixations``. Example::
247 {"trials": [{"participant_id": "p1", "trial_id": "t1", "parts": [
248 {"screen_id": "intro", "screen_index": 1,
249 "words": {"page": "intro"}, "fixations": {"page": "intro"}}
250 ]}]}
252 Selectors are exact column/value mappings. Overlapping selectors, unmatched
253 rows inside a declared parent, duplicate order keys, and unknown columns are
254 errors; the function never chooses a first row silently.
255 """
256 if normalized.empty:
257 return normalized
258 if kind not in {"words", "fixations"}:
259 raise ValueError("kind must be 'words' or 'fixations'.")
260 out = normalized.copy()
261 assigned = pd.Series(False, index=out.index)
262 declared_parent = pd.Series(False, index=out.index)
263 for trial in _manifest_trials(manifest):
264 if not isinstance(trial, Mapping):
265 raise ValueError("Each manifest trial must be an object.")
266 pid, tid = trial.get("participant_id"), trial.get("trial_id")
267 parts = trial.get("parts", trial.get("screens", []))
268 if pid is None or tid is None or not isinstance(parts, Sequence):
269 raise ValueError(
270 "Each manifest trial needs participant_id, trial_id, and parts."
271 )
272 parent_mask = (out["participant_id"].astype(str) == str(pid)) & (
273 out["trial_id"].astype(str) == str(tid)
274 )
275 if not parent_mask.any():
276 raise ValueError(
277 f"Manifest parent {(str(pid), str(tid))!r} matches no rows."
278 )
279 declared_parent |= parent_mask
280 for position, part in enumerate(parts, start=1):
281 if not isinstance(part, Mapping) or part.get(SCREEN_ID) in (None, ""):
282 raise ValueError("Each manifest part needs a non-empty screen_id.")
283 selector = part.get(kind)
284 if not isinstance(selector, Mapping) or not selector:
285 raise ValueError(
286 f"Manifest screen {part[SCREEN_ID]!r} needs a {kind} selector."
287 )
288 mask = parent_mask.copy()
289 for column, wanted in selector.items():
290 if column not in source.columns:
291 raise ValueError(
292 f"Manifest {kind} selector names unknown column {column!r}."
293 )
294 mask &= source[column].eq(wanted)
295 if not mask.any():
296 raise ValueError(
297 f"Manifest screen {part[SCREEN_ID]!r} {kind} selector matches no rows."
298 )
299 if (assigned & mask).any():
300 raise ValueError(
301 f"Manifest screen {part[SCREEN_ID]!r} overlaps another part."
302 )
303 assigned |= mask
304 out.loc[mask, SCREEN_ID] = str(part[SCREEN_ID])
305 out.loc[mask, SCREEN_INDEX] = int(part.get(SCREEN_INDEX, position))
306 for column in (CANVAS_WIDTH, CANVAS_HEIGHT):
307 if part.get(column) is not None:
308 out.loc[mask, column] = part[column]
309 if (declared_parent & ~assigned).any():
310 count = int((declared_parent & ~assigned).sum())
311 raise ValueError(
312 f"Trial-parts manifest leaves {count} declared-parent row(s) unmatched."
313 )
314 return normalize_screen_identity(out)
317def screen_canvas_size(frame: pd.DataFrame) -> tuple[int, int] | None:
318 """Per-screen canvas metadata when both dimensions are present.
320 ``None`` — the caller's own screen size — unless each dimension holds one
321 finite, positive value."""
322 if frame is None or frame.empty:
323 return None
324 values = []
325 for column in (CANVAS_WIDTH, CANVAS_HEIGHT):
326 if column not in frame.columns:
327 return None
328 numeric = pd.to_numeric(frame[column], errors="coerce").dropna().unique()
329 if len(numeric) != 1 or not np.isfinite(numeric[0]) or numeric[0] <= 0:
330 return None
331 values.append(int(numeric[0]))
332 return values[0], values[1]