Coverage for scanpath_studio/multipart.py: 98%

182 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Canonical identity and validation for trials made of ordered screens. 

2 

3Legacy Scanpath Studio data identifies one coordinate space with 

4``(participant_id, trial_id)``. Multipart trials retain that logical parent 

5identity and add ``screen_id`` plus a 1-based ``screen_index`` on every row. 

6This module is deliberately pandas-only so ingestion, measures, UI, API, CLI, 

7and export can share the exact same boundary rules without importing Streamlit. 

8""" 

9 

10from __future__ import annotations 

11 

12from collections.abc import Mapping, Sequence 

13from typing import Any 

14 

15import numpy as np 

16import pandas as pd 

17 

18PARENT_KEY = ("participant_id", "trial_id") 

19SCREEN_ID = "screen_id" 

20SCREEN_INDEX = "screen_index" 

21PART_KEY = (*PARENT_KEY, SCREEN_ID) 

22SCREEN_TIMESTAMP = "screen_timestamp_ms" 

23SCREEN_FIXATION_ID = "screen_fixation_id" 

24CANVAS_WIDTH = "canvas_width" 

25CANVAS_HEIGHT = "canvas_height" 

26 

27 

28def has_screen_identity(frame: pd.DataFrame | None) -> bool: 

29 """Whether ``frame`` carries canonical screen identity.""" 

30 return bool(frame is not None and SCREEN_ID in frame.columns) 

31 

32 

33def grouping_columns(frame: pd.DataFrame, *, include_word: bool = False) -> list[str]: 

34 """Identity columns for a scientific operation on ``frame``. 

35 

36 Single-screen frames keep the historical parent key. Multipart frames add 

37 ``screen_id`` so assignment, saccades, runs, passes, and measures can never 

38 connect two coordinate spaces accidentally. 

39 """ 

40 keys = [column for column in PARENT_KEY if column in frame.columns] 

41 if has_screen_identity(frame): 

42 keys.append(SCREEN_ID) 

43 if include_word and "word_id" in frame.columns: 

44 keys.append("word_id") 

45 return keys 

46 

47 

48def _parent_columns(frame: pd.DataFrame) -> list[str]: 

49 missing = [column for column in PARENT_KEY if column not in frame.columns] 

50 if missing: 

51 raise ValueError( 

52 "Multipart identity needs canonical parent columns: " + ", ".join(missing) 

53 ) 

54 return list(PARENT_KEY) 

55 

56 

57def normalize_screen_identity(frame: pd.DataFrame) -> pd.DataFrame: 

58 """Normalize/validate screen columns, preserving legacy frame shape. 

59 

60 ``screen_id`` alone is enough: order is derived from first appearance 

61 within each logical trial. ``screen_index`` alone is also accepted and its 

62 string value becomes the id. When both are supplied, the mapping must be 

63 one-to-one inside a parent. Canvas dimensions, when supplied, must be 

64 positive and constant within a screen. 

65 """ 

66 has_id = SCREEN_ID in frame.columns 

67 has_index = SCREEN_INDEX in frame.columns 

68 if not has_id and not has_index: 

69 return frame 

70 

71 parents = _parent_columns(frame) 

72 out = frame.copy() 

73 if not has_id: 

74 numeric_index = pd.to_numeric(out[SCREEN_INDEX], errors="coerce") 

75 if numeric_index.isna().any(): 

76 raise ValueError( 

77 "`screen_index` (screen order) has blank or non-numeric cells." 

78 ) 

79 out[SCREEN_ID] = numeric_index.astype(int).astype(str) 

80 else: 

81 # `stable_id`, not a plain `.astype(str)` — BUG-44's hazard applies here 

82 # too: a whole-number screen_id reads as float64 the moment any OTHER 

83 # row anywhere in that column is missing, so one report's `"1"` becomes 

84 # another's `"1.0"` and `validate_matching_parts` below rejects every 

85 # screen as an orphan even though both sides recorded the same one. 

86 # Imported locally — `data.py` imports from this module, so a 

87 # module-level import would cycle. 

88 from .data import stable_id 

89 

90 # A missing/blank cell is tolerated exactly as `trial_id_series` (via 

91 # `stable_id`) already tolerates one in a trial id — no proactive 

92 # `isna()` check here either. Rejecting it outright meant a screen_id 

93 # column with a stray blank cell (the same dtype-coercion quirk 

94 # BUG-44 fixed for identity columns generally) failed the whole 

95 # mapping instead of just reading that one blank cell as "nan". 

96 out[SCREEN_ID] = stable_id(out[SCREEN_ID]) 

97 

98 if not has_index: 

99 distinct = out[parents + [SCREEN_ID]].drop_duplicates() 

100 distinct[SCREEN_INDEX] = distinct.groupby(parents, sort=False).cumcount() + 1 

101 out = out.merge(distinct, on=parents + [SCREEN_ID], how="left", sort=False) 

102 else: 

103 numeric_index = pd.to_numeric(out[SCREEN_INDEX], errors="coerce") 

104 if numeric_index.isna().any() or (numeric_index <= 0).any(): 

105 raise ValueError("`screen_index` (screen order) must count from 1.") 

106 if (numeric_index % 1 != 0).any(): 

107 raise ValueError("`screen_index` (screen order) must be whole numbers.") 

108 out[SCREEN_INDEX] = numeric_index.astype(int) 

109 

110 pairs = out[parents + [SCREEN_ID, SCREEN_INDEX]].drop_duplicates() 

111 if pairs.duplicated(parents + [SCREEN_ID], keep=False).any(): 

112 raise ValueError("Within a trial, one Screen ID has two `screen_index` values.") 

113 if pairs.duplicated(parents + [SCREEN_INDEX], keep=False).any(): 

114 raise ValueError("Within a trial, two Screen IDs share one `screen_index`.") 

115 

116 for column in (CANVAS_WIDTH, CANVAS_HEIGHT): 

117 if column not in out.columns: 

118 continue 

119 values = pd.to_numeric(out[column], errors="coerce") 

120 if values.notna().any() and (values.dropna() <= 0).any(): 

121 raise ValueError(f"{_FIELD_LABELS[column]} must be positive.") 

122 out[column] = values 

123 counts = out.groupby(list(PART_KEY), dropna=False)[column].nunique(dropna=True) 

124 if (counts > 1).any(): 

125 raise ValueError(f"{_FIELD_LABELS[column]} changes within one screen.") 

126 return out 

127 

128 

129#: What a user calls each screen column in a message (the add screen's names). 

130_FIELD_LABELS = { 

131 SCREEN_INDEX: "`screen_index`", 

132 CANVAS_WIDTH: "Screen canvas width", 

133 CANVAS_HEIGHT: "Screen canvas height", 

134} 

135 

136 

137def _screens(parts: list) -> str: 

138 """Up to three ``(participant, trial, screen)`` keys, as ``p01/3/page_2``.""" 

139 return ", ".join("/".join(str(v) for v in part) for part in parts[:3]) 

140 

141 

142def part_catalog(*frames: pd.DataFrame | None) -> pd.DataFrame: 

143 """One ordered row per screen across ``frames``. 

144 

145 Metadata conflicts between words and fixations are rejected rather than 

146 resolved with a silent ``first()``. Legacy data returns an empty catalogue. 

147 """ 

148 rows: list[pd.DataFrame] = [] 

149 metadata = [SCREEN_INDEX, CANVAS_WIDTH, CANVAS_HEIGHT, "text_id"] 

150 for frame in frames: 

151 if frame is None or frame.empty or not has_screen_identity(frame): 

152 continue 

153 normalized = normalize_screen_identity(frame) 

154 columns = [*PART_KEY, *(c for c in metadata if c in normalized.columns)] 

155 candidate = normalized[columns].drop_duplicates() 

156 for column in metadata: 

157 if column not in candidate.columns: 

158 candidate[column] = pd.NA 

159 rows.append(candidate[[*PART_KEY, *metadata]]) 

160 if not rows: 

161 return pd.DataFrame(columns=[*PART_KEY, *metadata]) 

162 

163 combined = pd.concat(rows, ignore_index=True) 

164 for column in metadata: 

165 conflicts = combined.groupby(list(PART_KEY), dropna=False)[column].nunique( 

166 dropna=True 

167 ) 

168 if (conflicts > 1).any(): 

169 # tabs.py matches this text exactly (the screen_index case), so it 

170 # keeps its wording until that check reads something sturdier (#374). 

171 raise ValueError(f"Multipart metadata {column!r} conflicts across tables.") 

172 catalog = ( 

173 combined.groupby(list(PART_KEY), as_index=False, dropna=False) 

174 .first() 

175 .sort_values([*PARENT_KEY, SCREEN_INDEX, SCREEN_ID], kind="stable") 

176 .reset_index(drop=True) 

177 ) 

178 return catalog 

179 

180 

181def validate_matching_parts(words: pd.DataFrame, fixations: pd.DataFrame) -> None: 

182 """Reject orphan screens when both normalized reports carry part identity.""" 

183 if words.empty or fixations.empty: 

184 return 

185 if not has_screen_identity(words) and not has_screen_identity(fixations): 

186 return 

187 if has_screen_identity(words) != has_screen_identity(fixations): 

188 raise ValueError( 

189 "Screen ID is set in only one table; set it in both or neither." 

190 ) 

191 word_parts = set(map(tuple, words[list(PART_KEY)].drop_duplicates().to_numpy())) 

192 fixation_parts = set( 

193 map(tuple, fixations[list(PART_KEY)].drop_duplicates().to_numpy()) 

194 ) 

195 if word_parts != fixation_parts: 

196 missing_words = sorted(fixation_parts - word_parts) 

197 missing_fix = sorted(word_parts - fixation_parts) 

198 details = [] 

199 if missing_words: 

200 details.append(f"no words for {_screens(missing_words)}") 

201 if missing_fix: 

202 details.append(f"no fixations for {_screens(missing_fix)}") 

203 raise ValueError("Some screens are in one table only: " + "; ".join(details)) 

204 

205 

206def extract_part( 

207 frame: pd.DataFrame, 

208 participant_id: Any, 

209 trial_id: Any, 

210 screen_id: Any | None = None, 

211) -> pd.DataFrame: 

212 """Extract one parent trial or one screen without concatenating screens.""" 

213 if frame is None or frame.empty: 

214 return frame 

215 mask = (frame["participant_id"].astype(str) == str(participant_id)) & ( 

216 frame["trial_id"].astype(str) == str(trial_id) 

217 ) 

218 if screen_id is not None: 

219 if SCREEN_ID not in frame.columns: 

220 raise ValueError("screen= was supplied for a single-screen dataset.") 

221 mask &= frame[SCREEN_ID].astype(str) == str(screen_id) 

222 return frame.loc[mask] 

223 

224 

225def _manifest_trials(manifest: Mapping[str, Any] | Sequence[Mapping[str, Any]]) -> list: 

226 if isinstance(manifest, Mapping): 

227 trials = manifest.get("trials", manifest.get("multipart_trials", [])) 

228 else: 

229 trials = manifest 

230 if not isinstance(trials, Sequence) or isinstance(trials, (str, bytes)): 

231 raise ValueError("trial_parts_manifest must contain a 'trials' list.") 

232 return list(trials) 

233 

234 

235def apply_trial_parts_manifest( 

236 normalized: pd.DataFrame, 

237 source: pd.DataFrame, 

238 manifest: Mapping[str, Any] | Sequence[Mapping[str, Any]], 

239 *, 

240 kind: str, 

241) -> pd.DataFrame: 

242 """Attach a nested trial-parts manifest to a normalized report. 

243 

244 Each part declares ``screen_id``, optional ``screen_index``/canvas size, and 

245 a source-row selector under ``words`` or ``fixations``. Example:: 

246 

247 {"trials": [{"participant_id": "p1", "trial_id": "t1", "parts": [ 

248 {"screen_id": "intro", "screen_index": 1, 

249 "words": {"page": "intro"}, "fixations": {"page": "intro"}} 

250 ]}]} 

251 

252 Selectors are exact column/value mappings. Overlapping selectors, unmatched 

253 rows inside a declared parent, duplicate order keys, and unknown columns are 

254 errors; the function never chooses a first row silently. 

255 """ 

256 if normalized.empty: 

257 return normalized 

258 if kind not in {"words", "fixations"}: 

259 raise ValueError("kind must be 'words' or 'fixations'.") 

260 out = normalized.copy() 

261 assigned = pd.Series(False, index=out.index) 

262 declared_parent = pd.Series(False, index=out.index) 

263 for trial in _manifest_trials(manifest): 

264 if not isinstance(trial, Mapping): 

265 raise ValueError("Each manifest trial must be an object.") 

266 pid, tid = trial.get("participant_id"), trial.get("trial_id") 

267 parts = trial.get("parts", trial.get("screens", [])) 

268 if pid is None or tid is None or not isinstance(parts, Sequence): 

269 raise ValueError( 

270 "Each manifest trial needs participant_id, trial_id, and parts." 

271 ) 

272 parent_mask = (out["participant_id"].astype(str) == str(pid)) & ( 

273 out["trial_id"].astype(str) == str(tid) 

274 ) 

275 if not parent_mask.any(): 

276 raise ValueError( 

277 f"Manifest parent {(str(pid), str(tid))!r} matches no rows." 

278 ) 

279 declared_parent |= parent_mask 

280 for position, part in enumerate(parts, start=1): 

281 if not isinstance(part, Mapping) or part.get(SCREEN_ID) in (None, ""): 

282 raise ValueError("Each manifest part needs a non-empty screen_id.") 

283 selector = part.get(kind) 

284 if not isinstance(selector, Mapping) or not selector: 

285 raise ValueError( 

286 f"Manifest screen {part[SCREEN_ID]!r} needs a {kind} selector." 

287 ) 

288 mask = parent_mask.copy() 

289 for column, wanted in selector.items(): 

290 if column not in source.columns: 

291 raise ValueError( 

292 f"Manifest {kind} selector names unknown column {column!r}." 

293 ) 

294 mask &= source[column].eq(wanted) 

295 if not mask.any(): 

296 raise ValueError( 

297 f"Manifest screen {part[SCREEN_ID]!r} {kind} selector matches no rows." 

298 ) 

299 if (assigned & mask).any(): 

300 raise ValueError( 

301 f"Manifest screen {part[SCREEN_ID]!r} overlaps another part." 

302 ) 

303 assigned |= mask 

304 out.loc[mask, SCREEN_ID] = str(part[SCREEN_ID]) 

305 out.loc[mask, SCREEN_INDEX] = int(part.get(SCREEN_INDEX, position)) 

306 for column in (CANVAS_WIDTH, CANVAS_HEIGHT): 

307 if part.get(column) is not None: 

308 out.loc[mask, column] = part[column] 

309 if (declared_parent & ~assigned).any(): 

310 count = int((declared_parent & ~assigned).sum()) 

311 raise ValueError( 

312 f"Trial-parts manifest leaves {count} declared-parent row(s) unmatched." 

313 ) 

314 return normalize_screen_identity(out) 

315 

316 

317def screen_canvas_size(frame: pd.DataFrame) -> tuple[int, int] | None: 

318 """Per-screen canvas metadata when both dimensions are present. 

319 

320 ``None`` — the caller's own screen size — unless each dimension holds one 

321 finite, positive value.""" 

322 if frame is None or frame.empty: 

323 return None 

324 values = [] 

325 for column in (CANVAS_WIDTH, CANVAS_HEIGHT): 

326 if column not in frame.columns: 

327 return None 

328 numeric = pd.to_numeric(frame[column], errors="coerce").dropna().unique() 

329 if len(numeric) != 1 or not np.isfinite(numeric[0]) or numeric[0] <= 0: 

330 return None 

331 values.append(int(numeric[0])) 

332 return values[0], values[1]