Coverage for scanpath_studio/experimental_setup.py: 94%

148 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Display geometry and stimulus typography: conversions, and the per-dataset 

2setup a corpus was recorded with (DATA-2 · DATA-22 · CMP-8). 

3 

4Deliberately pure — no Streamlit, no pandas. That is what lets the headless API, 

5the wizard, the comparison figure and CMP-11's visual-angle mode all share one 

6notion of "what screen was this recorded on", and it keeps the type reachable 

7from `api.py` without importing `app`. 

8 

9Two things live here: 

10 

11* the conversions (`dpi_from_width`, `font_pt_to_px`, `pixels_per_degree`); 

12* `SetupSnapshot` — one dataset's canvas / physical size / typography, each of 

13 the three groups carrying a `Provenance` saying *how we know it*. 

14 

15The provenance is the point. An uploaded corpus that never recorded its monitor 

16used to silently inherit a 2560x1440 guess; a snapshot instead records that the 

17screen was ``ASSUMED``, and anything derived from a ``SKIPPED`` group 

18(px/degree, pt->px) resolves to ``None`` so a caller hides it rather than 

19printing a number computed from a default. 

20""" 

21 

22from __future__ import annotations 

23 

24import math 

25from collections.abc import Mapping 

26from dataclasses import dataclass 

27from enum import StrEnum 

28from typing import Any 

29 

30# Defaults duplicated from ``constants.py`` rather than imported: this module is 

31# the pure bottom of the dependency graph and must stay importable on its own. 

32_DEFAULT_CANVAS = (2560, 1440) 

33_DEFAULT_MONITOR_WIDTH_MM = 597.0 

34_DEFAULT_VIEWING_DISTANCE_MM = 800.0 

35_DEFAULT_BASE_FONT_SIZE = 16 

36_DEFAULT_FONT_FAMILY = "monospace" 

37_DEFAULT_LINE_SPACING = 3.0 

38 

39 

40def dpi_from_width(width_px: float, width_mm: float) -> float: 

41 """Horizontal display DPI from pixel and physical widths.""" 

42 if width_px <= 0 or width_mm <= 0: 

43 raise ValueError("display widths must be positive") 

44 return float(width_px) / (float(width_mm) / 25.4) 

45 

46 

47def font_pt_to_px(font_pt: float, dpi: float) -> float: 

48 """CSS/monitor pixels corresponding to a point size at ``dpi``.""" 

49 if font_pt <= 0 or dpi <= 0: 

50 raise ValueError("font size and DPI must be positive") 

51 return float(font_pt) * float(dpi) / 72.0 

52 

53 

54def pixels_per_degree( 

55 viewing_distance_mm: float, width_px: float, width_mm: float 

56) -> float: 

57 """Pixels subtending one visual degree at the configured setup.""" 

58 if viewing_distance_mm <= 0: 

59 raise ValueError("viewing distance must be positive") 

60 px_per_mm = float(width_px) / float(width_mm) 

61 mm_per_degree = 2.0 * float(viewing_distance_mm) * math.tan(math.radians(0.5)) 

62 return px_per_mm * mm_per_degree 

63 

64 

65class Provenance(StrEnum): 

66 """How a setup group's values came to be what they are. 

67 

68 ``StrEnum`` so a member serializes as its own value — the share param, the 

69 saved-config section and ``plot_config.json`` all write bare strings. 

70 """ 

71 

72 MEASURED = "measured" 

73 """The user knows the values, or the corpus declares them.""" 

74 ESTIMATED = "estimated" 

75 """Derived from the uploaded data (a lower bound, not the real screen).""" 

76 ASSUMED = "assumed" 

77 """A named default was taken.""" 

78 SKIPPED = "skipped" 

79 """Declined; everything derived from this group is hidden, not guessed.""" 

80 

81 

82#: The three groups a user answers in the wizard's Recording-setup step, in the 

83#: order they are asked. ``geometry`` is the only one that may be skipped. 

84SETUP_GROUPS: tuple[str, ...] = ("screen", "geometry", "text") 

85 

86#: Short names used on the wire (``setup_prov=screen:assumed,geom:skipped,...``). 

87#: Kept separate from ``SETUP_GROUPS`` so the param stays compact without 

88#: renaming the group everywhere else. 

89_GROUP_WIRE_NAMES: dict[str, str] = { 

90 "screen": "screen", 

91 "geometry": "geom", 

92 "text": "text", 

93} 

94_WIRE_NAME_GROUPS: dict[str, str] = {v: k for k, v in _GROUP_WIRE_NAMES.items()} 

95 

96SETUP_GROUP_LABELS: dict[str, str] = { 

97 "screen": "Screen", 

98 "geometry": "Physical size & viewing distance", 

99 "text": "Reading text size", 

100} 

101 

102 

103@dataclass(frozen=True) 

104class SetupSnapshot: 

105 """The screen and typography one dataset was set up with. 

106 

107 Every field is a resolved value — a snapshot never carries "unknown". What 

108 it carries instead is the per-group ``Provenance``, so a reader can tell a 

109 measured 1680x1050 from an assumed one and, for a ``SKIPPED`` group, decline 

110 to show the quantities that would otherwise be invented (`dpi`, 

111 `px_per_degree`, `stimulus_font_px`). 

112 """ 

113 

114 canvas_width: int = _DEFAULT_CANVAS[0] 

115 canvas_height: int = _DEFAULT_CANVAS[1] 

116 monitor_width_mm: float = _DEFAULT_MONITOR_WIDTH_MM 

117 viewing_distance_mm: float = _DEFAULT_VIEWING_DISTANCE_MM 

118 base_font_size: int = _DEFAULT_BASE_FONT_SIZE 

119 font_family: str = _DEFAULT_FONT_FAMILY 

120 line_spacing: float = _DEFAULT_LINE_SPACING 

121 scale_text_to_boxes: bool = True 

122 # Provenance is three scalar fields rather than a dict so the dataclass stays 

123 # hashable and cheap to compare (it rides inside cached values). 

124 screen_provenance: Provenance = Provenance.ASSUMED 

125 geometry_provenance: Provenance = Provenance.ASSUMED 

126 text_provenance: Provenance = Provenance.ASSUMED 

127 

128 # -- derived --------------------------------------------------------------- 

129 

130 @property 

131 def provenance(self) -> dict[str, Provenance]: 

132 """Group -> provenance, in ``SETUP_GROUPS`` order.""" 

133 return { 

134 "screen": self.screen_provenance, 

135 "geometry": self.geometry_provenance, 

136 "text": self.text_provenance, 

137 } 

138 

139 @property 

140 def dpi(self) -> float | None: 

141 """Horizontal DPI, or ``None`` when the physical size was skipped. 

142 

143 A DPI computed from a default monitor width is a guess wearing a 

144 number's clothes, so a skipped geometry group yields nothing at all. 

145 """ 

146 if self.geometry_provenance is Provenance.SKIPPED: 

147 return None 

148 try: 

149 return dpi_from_width(self.canvas_width, self.monitor_width_mm) 

150 except ValueError: 

151 return None 

152 

153 @property 

154 def px_per_degree(self) -> float | None: 

155 """Pixels per degree of visual angle, or ``None`` when skipped.""" 

156 if self.geometry_provenance is Provenance.SKIPPED: 

157 return None 

158 try: 

159 return pixels_per_degree( 

160 self.viewing_distance_mm, self.canvas_width, self.monitor_width_mm 

161 ) 

162 except (ValueError, ZeroDivisionError): 

163 return None 

164 

165 @property 

166 def canvas(self) -> tuple[int, int]: 

167 """``(width, height)`` — the shape the figure builders take.""" 

168 return (int(self.canvas_width), int(self.canvas_height)) 

169 

170 def stimulus_font_px(self, font_pt: float) -> float | None: 

171 """Convert a point size through this setup's DPI, or ``None`` if skipped.""" 

172 dpi = self.dpi 

173 if dpi is None: 

174 return None 

175 try: 

176 return font_pt_to_px(font_pt, dpi) 

177 except ValueError: 

178 return None 

179 

180 def is_answered(self) -> bool: 

181 """Every group carries a real provenance (the wizard's Add-dataset gate).""" 

182 return all(isinstance(p, Provenance) for p in self.provenance.values()) 

183 

184 # -- serialization --------------------------------------------------------- 

185 

186 def to_dict(self) -> dict[str, Any]: 

187 """JSON-able form written to the stored dataset payload, the saved-setup 

188 JSON, the recovery-cache manifest, and bulk export's ``plot_config.json``.""" 

189 return { 

190 "canvas_width": int(self.canvas_width), 

191 "canvas_height": int(self.canvas_height), 

192 "monitor_width_mm": float(self.monitor_width_mm), 

193 "viewing_distance_mm": float(self.viewing_distance_mm), 

194 "base_font_size": int(self.base_font_size), 

195 "font_family": str(self.font_family), 

196 "line_spacing": float(self.line_spacing), 

197 "scale_text_to_boxes": bool(self.scale_text_to_boxes), 

198 "provenance": {g: str(p) for g, p in self.provenance.items()}, 

199 } 

200 

201 @classmethod 

202 def from_dict( 

203 cls, 

204 payload: Mapping[str, Any] | None, 

205 *, 

206 fallback: SetupSnapshot | None = None, 

207 ) -> SetupSnapshot: 

208 """Read a snapshot back, degrading rather than raising. 

209 

210 ``fallback`` is required by the read path on purpose: a stored dataset or 

211 a recovery cache written before this key existed has no snapshot at all, 

212 and a corpus that cannot state its geometry must still open. Anything 

213 unparseable falls back field by field, so one bad number does not 

214 discard a whole valid setup. 

215 """ 

216 base = fallback if fallback is not None else cls() 

217 if not isinstance(payload, Mapping): 

218 return base 

219 

220 def _num(key: str, current, cast): 

221 try: 

222 value = payload[key] 

223 except (KeyError, TypeError): 

224 return current 

225 if value is None or isinstance(value, bool): 

226 return current 

227 try: 

228 return cast(value) 

229 except (TypeError, ValueError): 

230 return current 

231 

232 raw_prov = payload.get("provenance") 

233 prov = dict(base.provenance) 

234 if isinstance(raw_prov, Mapping): 

235 for group in SETUP_GROUPS: 

236 parsed = _coerce_provenance(raw_prov.get(group)) 

237 if parsed is not None: 

238 prov[group] = parsed 

239 

240 scale = payload.get("scale_text_to_boxes") 

241 return cls( 

242 canvas_width=_num("canvas_width", base.canvas_width, int), 

243 canvas_height=_num("canvas_height", base.canvas_height, int), 

244 monitor_width_mm=_num("monitor_width_mm", base.monitor_width_mm, float), 

245 viewing_distance_mm=_num( 

246 "viewing_distance_mm", base.viewing_distance_mm, float 

247 ), 

248 base_font_size=_num("base_font_size", base.base_font_size, int), 

249 font_family=str(payload.get("font_family") or base.font_family), 

250 line_spacing=_num("line_spacing", base.line_spacing, float), 

251 scale_text_to_boxes=( 

252 bool(scale) if isinstance(scale, bool) else base.scale_text_to_boxes 

253 ), 

254 screen_provenance=prov["screen"], 

255 geometry_provenance=prov["geometry"], 

256 text_provenance=prov["text"], 

257 ) 

258 

259 

260def _coerce_provenance(value: Any) -> Provenance | None: 

261 """A ``Provenance`` from a wire string, or ``None`` when unrecognised.""" 

262 if isinstance(value, Provenance): 

263 return value 

264 if not isinstance(value, str): 

265 return None 

266 try: 

267 return Provenance(value.strip().lower()) 

268 except ValueError: 

269 return None 

270 

271 

272class IncomparableScreensError(ValueError): 

273 """An overlay was asked of two readings recorded on different screens. 

274 

275 Raised by `api.compare_scanpaths`, which draws nothing rather than switch 

276 layout behind a script's back, and by `api.animate_scanpath` for a 

277 co-animation of two datasets, which is an overlay on one clock (CMP-21). 

278 ``reason`` is `setups_comparable`'s surface-neutral sentence, so a caller 

279 can word the way out in its own terms — `render` names ``--compare-layout`` 

280 rather than echo the Python keyword. 

281 """ 

282 

283 def __init__(self, message: str, *, reason: str) -> None: 

284 super().__init__(message) 

285 self.reason = reason 

286 

287 

288#: Provenance values that mean "we know what screen this was" — the corpus said 

289#: so, or it was inferred from the data. `ASSUMED` is excluded on purpose: it 

290#: means a named default was taken, and two datasets that both defaulted are two 

291#: unknowns rather than a known-equal. 

292_REAL_SCREEN_PROVENANCE = frozenset({Provenance.MEASURED, Provenance.ESTIMATED}) 

293 

294 

295def setups_comparable(a: SetupSnapshot, b: SetupSnapshot) -> tuple[bool, str]: 

296 """Whether two datasets' readings can share one set of pixel coordinates (CMP-11). 

297 

298 Compare mode's **overlay** layout pools both trials into one axis range, so 

299 it is only meaningful when a pixel means the same thing on both sides. CMP-8 

300 handled that by refusing every cross-dataset pair; this is the finer question 

301 it was standing in for. 

302 

303 Returns ``(allowed, note)``: 

304 

305 * ``(False, reason)`` — the canvases differ. The overlay is refused, and 

306 ``reason`` says why. 

307 * ``(True, caution)`` — the canvases match, but at least one side never 

308 *recorded* its screen (`screen_provenance` outside 

309 ``_REAL_SCREEN_PROVENANCE``), so the match may be a coincidence of 

310 defaults. The overlay is drawn and ``caution`` is surfaced beside it. 

311 * ``(True, "")`` — the canvases match and both sides know their screen. 

312 

313 ``note`` is a complete user-facing sentence in both non-empty cases, and the 

314 app, the CLI, :func:`api.compare_scanpaths` and :func:`api.animate_scanpath` 

315 all quote it whole, so the explanation cannot drift across surfaces. A 

316 refusal says only *why*: what happens next differs by surface — the app 

317 falls back to side by side, the API raises :class:`IncomparableScreensError`, 

318 ``render`` exits naming its own flag — so each caller appends that itself 

319 (BUG-85; the reason used to end "so they are shown side by side instead", 

320 which was false on two of three). 

321 

322 **Only the canvas is a hard gate.** An unrecorded screen warns rather than 

323 refuses — settled 2026-08-12 on the case that motivated it: two OneStop 

324 regimes, both 2560x1440, both reporting ``ASSUMED`` because the corpus does 

325 not record a screen. Refusing there blocked precisely the comparison the 

326 feature exists for. A caution is the honest middle: the app cannot *prove* 

327 the two displays matched, but the user usually can, and a matching canvas is 

328 real evidence rather than none. 

329 

330 **Physical geometry is deliberately not consulted.** Nothing in the overlay 

331 path converts to degrees — CMP-11 shipped as a gate, not a rescaling — so 

332 ``monitor_width_mm`` and ``viewing_distance_mm`` are never read, and 

333 requiring them to match would gate the feature on quantities it does not use. 

334 It would also make it inert: every built-in corpus hard-codes 

335 ``geometry: ASSUMED``, and only the wizard's "I know my monitor" branch ever 

336 reaches ``MEASURED``. 

337 

338 The residual cost is real and is disclosed rather than hidden: two corpora at 

339 1920x1080 on a 24-inch and a 32-inch monitor compare equal here, with only 

340 the caution to say so. The figure claims nothing but pixel positions, and its 

341 caption names both screens. 

342 """ 

343 if a.canvas != b.canvas: 

344 return False, ( 

345 f"These trials were recorded on different screens — " 

346 f"{a.canvas_width}×{a.canvas_height} and " 

347 f"{b.canvas_width}×{b.canvas_height} px — so their positions can't " 

348 f"share axes." 

349 ) 

350 unknown = [ 

351 name 

352 for name, snapshot in (("A", a), ("B", b)) 

353 if snapshot.screen_provenance not in _REAL_SCREEN_PROVENANCE 

354 ] 

355 if unknown: 

356 subject = ( 

357 "Neither dataset records" 

358 if len(unknown) == 2 

359 else f"Scanpath {unknown[0]}'s dataset does not record" 

360 ) 

361 return True, ( 

362 f"{subject} its screen, so the matching " 

363 f"{a.canvas_width}×{a.canvas_height} canvas may be a default, not " 

364 f"proof both were shown on one display. Check before reading " 

365 f"positions across the overlay." 

366 ) 

367 return True, "" 

368 

369 

370def format_provenance_param(snapshot: SetupSnapshot) -> str: 

371 """The compact ``setup_prov`` share value, e.g. 

372 ``screen:assumed,geom:skipped,text:measured``. 

373 

374 Provenance is metadata *about* settings rather than a setting, so it travels 

375 only far enough to stop a recipient being misled about numbers they did not 

376 choose — it changes no figure and takes no input. 

377 """ 

378 return ",".join( 

379 f"{_GROUP_WIRE_NAMES[group]}:{snapshot.provenance[group]}" 

380 for group in SETUP_GROUPS 

381 ) 

382 

383 

384def parse_provenance_param(value: str | None) -> dict[str, Provenance]: 

385 """Parse ``setup_prov`` into ``{group: Provenance}``. 

386 

387 Tolerant by contract — it reads a URL a stranger may have edited: unknown 

388 group names and unknown provenance words are dropped rather than raising, so 

389 a mangled param degrades to "no badges" instead of breaking the link. 

390 """ 

391 out: dict[str, Provenance] = {} 

392 if not value or not isinstance(value, str): 

393 return out 

394 for chunk in value.split(","): 

395 name, _, raw = chunk.partition(":") 

396 group = _WIRE_NAME_GROUPS.get(name.strip().lower()) 

397 if group is None: 

398 continue 

399 parsed = _coerce_provenance(raw) 

400 if parsed is not None: 

401 out[group] = parsed 

402 return out