Coverage for scanpath_studio/computations.py: 100%

102 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""The computation register (VAL-5). 

2 

3Every operation that **derives or semantically changes a value** a user, an 

4export or an API consumer can see, recorded once with its formula, its units, 

5its missing-data behaviour, and how far it has actually been verified. Pure UI 

6layout and byte-preserving file I/O are out of scope; filtering, precedence and 

7assignment are in, because they change *which observations* a result stands for 

8even when no arithmetic happens. 

9 

10This module is data, not behaviour. It exists so that 

11 

12* ``tests/test_computations.py`` can assert the catalogue has not drifted from 

13 the code — every ``aggregation.MEASURES`` entry, every drift-correction 

14 algorithm and every similarity metric must map to an entry here, and every 

15 entry must point at a function that exists; 

16* ``docs/computations.md`` can be generated from one source rather than 

17 maintained in parallel with it (``python -m scanpath_studio.computations``). 

18 

19**Verification tiers.** ``A`` a hand-calculated synthetic oracle · ``B`` an 

20independent reference implementation or corpus comparison · ``C`` property / 

21invariant tests · ``D`` cross-surface parity (UI, API, CLI, export agree). 

22 

23**Status is deliberately conservative.** *Verified* means a semantic oracle 

24exists and passes — not that a line of code was executed. Tier B is 

25systematically absent: it needs an independent implementation to compare 

26against, which is #VAL-4, on hold at the user's request until the register 

27itself has been read. Every scientific measure therefore reads *Partially 

28verified* even where its hand oracle is exact, and that is the honest state of 

29the world rather than a gap to paper over. *Convention* marks a choice that 

30cannot be right or wrong, only documented — a display transform, a tie-break, a 

31default threshold. 

32""" 

33 

34from __future__ import annotations 

35 

36import re 

37from dataclasses import dataclass, field 

38 

39#: Bumped when an entry's *meaning* changes (a formula, a unit, a default), not 

40#: when prose is edited. Exported alongside results so a bundle can name the 

41#: methodology it was produced under. 

42REGISTER_VERSION = "2" 

43 

44CATEGORY_IMPORTED = "Imported / precomputed" 

45CATEGORY_NORMALIZATION = "Normalization / inference" 

46CATEGORY_ASSIGNMENT = "Assignment / classification" 

47CATEGORY_PREPROCESSING = "Preprocessing" 

48CATEGORY_MEASURE = "Scientific measure" 

49CATEGORY_AGGREGATION = "Statistical aggregation" 

50CATEGORY_SIMILARITY = "Similarity" 

51CATEGORY_GEOMETRY = "Unit / coordinate conversion" 

52CATEGORY_DISPLAY = "Display / export transformation" 

53 

54CATEGORIES: tuple[str, ...] = ( 

55 CATEGORY_IMPORTED, 

56 CATEGORY_NORMALIZATION, 

57 CATEGORY_ASSIGNMENT, 

58 CATEGORY_PREPROCESSING, 

59 CATEGORY_MEASURE, 

60 CATEGORY_AGGREGATION, 

61 CATEGORY_SIMILARITY, 

62 CATEGORY_GEOMETRY, 

63 CATEGORY_DISPLAY, 

64) 

65 

66STATUS_VERIFIED = "Verified" 

67STATUS_PARTIAL = "Partially verified" 

68STATUS_UNVERIFIED = "Unverified" 

69STATUS_CONVENTION = "Intentional convention" 

70 

71STATUSES: tuple[str, ...] = ( 

72 STATUS_VERIFIED, 

73 STATUS_PARTIAL, 

74 STATUS_UNVERIFIED, 

75 STATUS_CONVENTION, 

76) 

77 

78 

79@dataclass(frozen=True) 

80class Computation: 

81 """One derived value, with everything needed to reproduce and judge it.""" 

82 

83 id: str 

84 name: str 

85 category: str 

86 summary: str 

87 formula: str 

88 code: str 

89 output: str = "" 

90 unit: str = "" 

91 grouping: str = "" 

92 missing: str = "" 

93 precedence: str = "" 

94 tiers: str = "" 

95 status: str = STATUS_UNVERIFIED 

96 reference: str = "" 

97 consumers: tuple[str, ...] = field(default_factory=tuple) 

98 tests: tuple[str, ...] = field(default_factory=tuple) 

99 

100 @property 

101 def module(self) -> str: 

102 return self.code.split(":", 1)[0] 

103 

104 @property 

105 def symbol(self) -> str: 

106 return self.code.split(":", 1)[1] if ":" in self.code else "" 

107 

108 

109_UI = "UI" 

110_API = "API" 

111_CLI = "CLI" 

112_EXPORT = "Export" 

113_CORPUS = "Corpus Analysis" 

114_INSPECT = "Data Management" 

115# Surfaces held back from the app this release behind `SCANPATH_EXPERIMENTAL=1`, 

116# named as such so the register does not advertise a panel a user cannot open. 

117_UI_PREPROCESSING = "UI (Preprocessing panel — not in this release, PRE-22)" 

118_INSPECT_DERIVED = "Data Management (derived tables — not in this release, UX-126)" 

119_API_EXPERIMENTAL = "API (not in this release, PRE-21)" 

120# FFD / FPRT / RPD / single-fixation duration, on both paths: computed 

121# (`measures.compute_per_word_measures`) and imported 

122# (`data._blank_unfixated_measures`, #374). 

123_UNFIXATED_MISSING = ( 

124 "Never fixated ⇒ NaN, not 0, so a skipped word is left out of every mean. " 

125 "An imported 0 is blanked too, wherever the word's fixation count is 0 — " 

126 "or, with no count mapped, its total fixation duration is 0 (BUG-63)." 

127) 

128 

129 

130REGISTER: tuple[Computation, ...] = ( 

131 # ------------------------------------------------------------------ 

132 # Normalization / inference 

133 # ------------------------------------------------------------------ 

134 Computation( 

135 id="norm.words", 

136 name="Word table normalization", 

137 category=CATEGORY_NORMALIZATION, 

138 summary="Map an arbitrary word/IA export onto the canonical word columns.", 

139 formula=( 

140 "For each canonical field, `pick_column` walks a candidate list and " 

141 "takes the first column that exists; the user's mapping overrides it. " 

142 "Unmapped optional fields are dropped unless listed in " 

143 "`WORD_OPTIONAL_FIELDS`." 

144 ), 

145 code="scanpath_studio/data.py:normalize_words", 

146 output="participant_id, trial_id, text_id, word_id, text, x, y, width, height", 

147 precedence="An explicit user mapping always beats auto-detection.", 

148 missing="A missing *required* field raises with the columns it looked for.", 

149 tiers="C, D", 

150 status=STATUS_PARTIAL, 

151 consumers=(_UI, _API, _CLI, _EXPORT), 

152 tests=("tests/test_data.py", "tests/test_column_mapping.py"), 

153 ), 

154 Computation( 

155 id="norm.fixations", 

156 name="Fixation table normalization", 

157 category=CATEGORY_NORMALIZATION, 

158 summary="Map an arbitrary fixation report onto the canonical columns.", 

159 formula=( 

160 "As `norm.words`, over the fixation candidate lists. `order_in_trial` " 

161 "is assigned by sorting each trial on `timestamp_ms`; `fixation_id` is " 

162 "synthesized per trial when the export carries none." 

163 ), 

164 code="scanpath_studio/data.py:normalize_fixations", 

165 output="participant_id, trial_id, x, y, duration_ms, timestamp_ms, …", 

166 grouping="(participant_id, trial_id[, screen_id])", 

167 missing="Rows with no coordinates survive when a word/AoI id is mapped.", 

168 tiers="C, D", 

169 status=STATUS_PARTIAL, 

170 consumers=(_UI, _API, _CLI, _EXPORT), 

171 tests=("tests/test_data.py",), 

172 ), 

173 Computation( 

174 id="norm.box_edges", 

175 name="Word box from edges", 

176 category=CATEGORY_NORMALIZATION, 

177 summary="Convert EyeLink IA edges to origin+size.", 

178 formula=( 

179 "x = IA_LEFT · y = IA_TOP · width = IA_RIGHT − IA_LEFT · " 

180 "height = IA_BOTTOM − IA_TOP." 

181 ), 

182 code="scanpath_studio/data.py:normalize_words", 

183 output="x, y, width, height", 

184 unit="px (screen coordinates, y increasing downwards)", 

185 tiers="A, C", 

186 status=STATUS_VERIFIED, 

187 consumers=(_UI, _API, _CLI, _EXPORT), 

188 tests=("tests/test_word_box_geometry.py",), 

189 ), 

190 Computation( 

191 id="norm.trial_id_composite", 

192 name="Composite trial identity", 

193 category=CATEGORY_NORMALIZATION, 

194 summary="Build one unique trial id from several columns.", 

195 formula=( 

196 "The mapped Trial ID columns are joined in the order given, " 

197 "separated by `_`, after casting each to string. A `_` or `\\` " 

198 "inside a part is escaped with a `\\` first, so two different " 

199 "tuples never give the same id; parts with neither compose as a " 

200 "plain join." 

201 ), 

202 code="scanpath_studio/data.py:trial_id_series", 

203 output="trial_id", 

204 missing="A row missing any component keeps the literal string of that part.", 

205 tiers="C, D", 

206 status=STATUS_PARTIAL, 

207 consumers=(_UI, _API, _CLI), 

208 tests=("tests/test_trial_identity.py", "tests/test_composite_ids.py"), 

209 ), 

210 Computation( 

211 id="norm.flags", 

212 name="Flag coercion", 

213 category=CATEGORY_NORMALIZATION, 

214 summary="Read EyeLink's string booleans as booleans (BUG-7).", 

215 formula=( 

216 "Numbers go by `!= 0`. Strings are matched case-insensitively " 

217 "against `{'', '.', '0', '0.0', 'false', 'f', 'no', 'n', 'na', " 

218 "'nan', '-'}` → False; anything else → True." 

219 ), 

220 code="scanpath_studio/data.py:coerce_flag", 

221 output="bool", 

222 missing=( 

223 "NaN → False for an operational flag (blink, excluded). A supplied " 

224 "reading-measure flag (skip, regression in / out) keeps it missing " 

225 "instead — `''`, `'.'`, `'na'`, `'nan'`, `'-'` and NaN read as NA " 

226 "(`coerce_measure_flag`, a nullable boolean)." 

227 ), 

228 tiers="A, C", 

229 status=STATUS_VERIFIED, 

230 reference="Guards the `'.'`-as-missing convention in EyeLink IA reports.", 

231 consumers=(_UI, _API, _CLI, _EXPORT), 

232 tests=("tests/test_data.py",), 

233 ), 

234 Computation( 

235 id="norm.stimulus_broadcast", 

236 name="Stimulus-level word broadcast", 

237 category=CATEGORY_NORMALIZATION, 

238 summary="Share one stimulus' word boxes across every participant who read it.", 

239 formula=( 

240 "Words with no participant column are copied once per reading " 

241 "(participant × trial [× screen]) in the fixations, stamped with " 

242 "that reading's ids. Per reading, the boxes are those of the first " 

243 "words trial found by its trial ID, then its trial ID before a " 

244 "repeat's _r2 suffix, then its Text ID (only a Text ID the fixations " 

245 "map, and never one the words give to more than one trial). A " 

246 "trial-ID match always stands; mapped Text IDs that disagree with " 

247 "it are warned about (DATA-49)." 

248 ), 

249 code="scanpath_studio/data.py:broadcast_stimulus_words", 

250 missing=( 

251 "No fixations for a text ⇒ its words are not broadcast. Some " 

252 "readings unmatched ⇒ StimulusJoinWarning with the counts; none " 

253 "matched, or any multipart screen unmatched ⇒ StimulusJoinError, " 

254 "never an empty table." 

255 ), 

256 tiers="C", 

257 status=STATUS_PARTIAL, 

258 consumers=(_UI, _API, _CLI), 

259 tests=("tests/test_stimulus_join.py", "tests/test_dataset_support.py"), 

260 ), 

261 Computation( 

262 id="norm.aoi_center_placement", 

263 name="AoI-only fixation placement", 

264 category=CATEGORY_NORMALIZATION, 

265 summary="Place a fixation with no x/y at its word box's center.", 

266 formula="x = word.x + width/2 · y = word.y + height/2.", 

267 code="scanpath_studio/data.py:harmonize_frames", 

268 unit="px", 

269 precedence="Only when x/y are absent; recorded coordinates always win.", 

270 missing="No matching word box ⇒ the fixation keeps no coordinates.", 

271 tiers="A, C", 

272 status=STATUS_VERIFIED, 

273 consumers=(_UI, _API, _CLI), 

274 tests=("tests/test_data.py",), 

275 ), 

276 Computation( 

277 id="norm.participant_metadata", 

278 name="Participant metadata join", 

279 category=CATEGORY_NORMALIZATION, 

280 summary="Attach a participant-level table without broadcasting it (DATA-20).", 

281 formula=( 

282 "Left join on string `participant_id`. Duplicate ids that agree " 

283 "are combined field by field (each field keeps the one non-missing " 

284 "value the rows hold); duplicate ids that **disagree** are dropped " 

285 "and reported, " 

286 "so no `groupby.first()` winner is ever invented. A field is " 

287 "projected onto the per-trial frame, never onto word/fixation rows." 

288 ), 

289 code="scanpath_studio/metadata.py:build_participant_metadata", 

290 output="One column per registered field, at participant grain", 

291 missing="A participant with no row reads as missing everywhere, never as a default.", 

292 precedence="A real recorded column of the same name always wins.", 

293 tiers="A, C, D", 

294 status=STATUS_VERIFIED, 

295 consumers=(_UI, _API, _CLI, _EXPORT, _INSPECT), 

296 tests=("tests/test_metadata.py", "tests/test_metadata_duplicates.py"), 

297 ), 

298 # ------------------------------------------------------------------ 

299 # Assignment / classification 

300 # ------------------------------------------------------------------ 

301 Computation( 

302 id="assign.fixation_to_word", 

303 name="Fixation → word assignment", 

304 category=CATEGORY_ASSIGNMENT, 

305 summary="The single highest-risk step: which word a fixation counts for.", 

306 formula=( 

307 "1. Bounding-box containment against the trial's word boxes — the " 

308 "experiment's own rectangles (`geom.word_box_bounds`), so on a " 

309 "tiling corpus a fixation on the space *after* a word is credited to " 

310 "that word, as EyeLink's interest-area report credits it. Boxes are " 

311 "half-open, `x0 ≤ x < x1` and `y0 ≤ y < y1` " 

312 "(`measures.word_box_contains`), so a point on an edge two boxes " 

313 "share goes to the one that starts there — the next word, the line " 

314 "below — as EyeLink assigns it. " 

315 "2. Otherwise `word_id = NaN` (out of text). There is no snapping " 

316 "to a nearby word." 

317 ), 

318 code="scanpath_studio/measures.py:assign_fixations_to_words", 

319 output="word_id", 

320 grouping="(participant_id, trial_id[, screen_id]) — never across screens", 

321 missing="Unassignable fixations keep NaN and are excluded from word measures.", 

322 precedence=( 

323 "Runs only when the fixations carry no word id. A mapped `word_id` " 

324 "(on the bundled demo, EyeLink's `CURRENT_FIX_INTEREST_AREA_ID`) is " 

325 "used exactly as given, blanks included — nothing is computed and " 

326 "no blank is filled — unless `overwrite=True`. #BUG-83: geometry " 

327 "agrees with that column on all 3,208 of the demo's EyeLink-assigned " 

328 "fixations." 

329 ), 

330 tiers="A, C", 

331 status=STATUS_PARTIAL, 

332 consumers=(_UI, _API, _CLI, _EXPORT, _CORPUS), 

333 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

334 ), 

335 Computation( 

336 id="assign.in_text", 

337 name="Out-of-text flag", 

338 category=CATEGORY_ASSIGNMENT, 

339 summary="Whether a fixation landed on any word of the stimulus.", 

340 formula=( 

341 "The fixation falls inside some word box (`word_box_bounds`, tested " 

342 "half-open by `word_box_contains`, as `assign.fixation_to_word` " 

343 "tests it). Box containment only, so a fixation the data's own " 

344 "`word_id` puts on a word but that lies outside every box still " 

345 "counts as out-of-text." 

346 ), 

347 code="scanpath_studio/measures.py:fixation_in_text_mask", 

348 output="bool mask", 

349 tiers="A, C", 

350 status=STATUS_VERIFIED, 

351 consumers=(_UI, _API, _CORPUS), 

352 tests=("tests/test_synthetic.py",), 

353 ), 

354 Computation( 

355 id="assign.line_cluster", 

356 name="Visual line clustering", 

357 category=CATEGORY_ASSIGNMENT, 

358 summary="Derive text lines from word-box geometry, not from `line_idx`.", 

359 formula=( 

360 "Word boxes are sorted by `y` and split wherever the gap between " 

361 "consecutive centers exceeds `tol_frac` (0.5) of the median box " 

362 "height. Exists because `line_idx` is a constant in many IA exports." 

363 ), 

364 code="scanpath_studio/measures.py:cluster_word_lines", 

365 output="Line index per word", 

366 tiers="A, C", 

367 status=STATUS_PARTIAL, 

368 consumers=(_UI, _API, _CORPUS), 

369 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

370 ), 

371 Computation( 

372 id="assign.runs", 

373 name="Runs and passes", 

374 category=CATEGORY_ASSIGNMENT, 

375 summary="Trial run, line run, and per-word visit/pass indices (PRE-16).", 

376 formula=( 

377 "Consecutive fixations on the same word form one *visit*; the n-th " 

378 "visit to a word is its n-th pass. Line runs break whenever the " 

379 "assigned line changes." 

380 ), 

381 code="scanpath_studio/measures.py:materialize_runs", 

382 output=( 

383 "run, linerun, word_runid, word_run (the visit's pass number), " 

384 "word_run_fix, nrun, reread (word_run > 1)" 

385 ), 

386 grouping="Ordered by `timestamp_ms` within a trial", 

387 precedence=( 

388 "Always recomputed: an imported column under any of these names is " 

389 "replaced. An imported `pass_index` (EyeLink's `reread` is renamed " 

390 "to it on load) is a separate column and is kept as given — " 

391 "nothing computes `pass_index`." 

392 ), 

393 tiers="A, C", 

394 status=STATUS_PARTIAL, 

395 consumers=(_UI, _API, _EXPORT, _CORPUS), 

396 tests=("tests/test_measures.py",), 

397 ), 

398 Computation( 

399 id="assign.progression", 

400 name="Progression and regression flags", 

401 category=CATEGORY_ASSIGNMENT, 

402 summary="Whether the *outgoing* saccade moves forward in the text.", 

403 formula=( 

404 "`progression = sign(next word_id − word_id)`. " 

405 "`is_regression = word_id < running max word_id in the trial` — i.e. " 

406 "relative to the furthest word reached, not to the previous fixation." 

407 ), 

408 code="scanpath_studio/measures.py:enrich_fixations", 

409 output="progression ∈ {−1, 0, 1}, is_regression", 

410 grouping="Per trial, in timestamp order", 

411 missing="Unassigned fixations give progression 0.", 

412 tiers="A, C", 

413 status=STATUS_VERIFIED, 

414 consumers=(_UI, _API, _EXPORT, _CORPUS), 

415 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

416 ), 

417 Computation( 

418 id="assign.saccade_class", 

419 name="Saccade reading class", 

420 category=CATEGORY_ASSIGNMENT, 

421 summary="Label each outgoing saccade by its reading role (VIZ-8).", 

422 formula=( 

423 "From the word and text line of the two fixations, in this order: " 

424 "refixation (same word), regression (up to an earlier line, or back " 

425 "within a line), return sweep (down to a later line), forward (the " 

426 "next word on the line), skip (two or more words ahead on the line); " 

427 "`other` when either fixation has no assigned word." 

428 ), 

429 code="scanpath_studio/measures.py:classify_saccades", 

430 output="(not stored — computed for each figure)", 

431 precedence=( 

432 "Always computed; an imported `saccade_type` / `NEXT_SAC_DIRECTION` " 

433 "(a direction) is not used." 

434 ), 

435 tiers="A, C", 

436 status=STATUS_PARTIAL, 

437 consumers=(_UI, _API, _EXPORT), 

438 tests=("tests/test_saccade_class_filter.py",), 

439 ), 

440 # ------------------------------------------------------------------ 

441 # Scientific measures 

442 # ------------------------------------------------------------------ 

443 Computation( 

444 id="measure.ffd", 

445 name="First fixation duration (FFD)", 

446 category=CATEGORY_MEASURE, 

447 summary="Duration of the first fixation on a word.", 

448 formula=( 

449 "Duration of the word's first fixation, whenever it comes — as " 

450 "EyeLink's `IA_FIRST_FIXATION_DURATION`, so a computed and an " 

451 "imported value mean the same. Not conditioned on first pass: a word " 

452 "first reached by a regression has an FFD and `skip_flag = True`; " 

453 "filter on `skip_flag` for first-pass-only analyses." 

454 ), 

455 code="scanpath_studio/measures.py:compute_per_word_measures", 

456 output="first_fixation_ms", 

457 unit="ms", 

458 grouping="(participant, trial, word)", 

459 missing=_UNFIXATED_MISSING, 

460 precedence="A precomputed `IA_FIRST_FIXATION_DURATION` wins.", 

461 tiers="A, D", 

462 status=STATUS_PARTIAL, 

463 reference="Rayner (1998), standard reading-measure definitions.", 

464 consumers=(_UI, _API), 

465 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

466 ), 

467 Computation( 

468 id="measure.fprt", 

469 name="First-pass gaze duration (FPRT)", 

470 category=CATEGORY_MEASURE, 

471 summary="Sum of the fixations in the word's first visit.", 

472 formula=( 

473 "Sum of every fixation in the word's **first** run, i.e. before the " 

474 "gaze leaves the word for the first time — whenever that run starts " 

475 "(EyeLink's `IA_FIRST_RUN_DWELL_TIME`; not conditioned on first pass, " 

476 "as `measure.ffd`). A fixation outside every word ends the run " 

477 "(BUG-66), as it does for `measure.second_pass`." 

478 ), 

479 code="scanpath_studio/measures.py:compute_per_word_measures", 

480 output="first_pass_gaze_duration_ms", 

481 unit="ms", 

482 grouping="(participant, trial, word)", 

483 missing=_UNFIXATED_MISSING, 

484 precedence="A precomputed IA gaze duration wins.", 

485 tiers="A, D", 

486 status=STATUS_PARTIAL, 

487 reference="Rayner (1998).", 

488 consumers=(_UI, _API), 

489 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

490 ), 

491 Computation( 

492 id="measure.rpd", 

493 name="Regression-path duration (RPD / go-past)", 

494 category=CATEGORY_MEASURE, 

495 summary="First entry to the word until the gaze passes it to the right.", 

496 formula=( 

497 "Total time from the word's first fixation until the first fixation " 

498 "on a **later** word — every fixation in between, including a first " 

499 "visit to an earlier, skipped word during the regression (BUG-61). " 

500 "Matches EyeLink's `IA_REGRESSION_PATH_DURATION` on 1779 of the " 

501 "bundled demo's 1780 fixated words. Fixations outside every word " 

502 "neither extend nor close the window." 

503 ), 

504 code="scanpath_studio/measures.py:compute_per_word_measures", 

505 output="regression_path_duration_ms", 

506 unit="ms", 

507 grouping="(participant, trial, word)", 

508 missing=_UNFIXATED_MISSING, 

509 tiers="A", 

510 status=STATUS_PARTIAL, 

511 reference=( 

512 "Definitions differ across toolkits (go-past vs regression path); " 

513 "`eyekit` is the intended comparison. Unresolved until that " 

514 "cross-validation runs." 

515 ), 

516 consumers=(_UI, _API), 

517 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

518 ), 

519 Computation( 

520 id="measure.tfd", 

521 name="Total fixation duration (TFD)", 

522 category=CATEGORY_MEASURE, 

523 summary="All time spent on a word across the whole trial.", 

524 formula="Sum of every fixation assigned to the word, any pass.", 

525 code="scanpath_studio/measures.py:compute_per_word_measures", 

526 output="total_fixation_duration_ms", 

527 unit="ms", 

528 missing="Never fixated ⇒ 0 (the word *was* read past; it got no time).", 

529 precedence="A precomputed IA dwell time wins.", 

530 tiers="A, D", 

531 status=STATUS_PARTIAL, 

532 consumers=(_UI, _API), 

533 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

534 ), 

535 Computation( 

536 id="measure.nfix", 

537 name="Fixations per word", 

538 category=CATEGORY_MEASURE, 

539 summary="Count of fixations assigned to a word.", 

540 formula="Row count of the word's assigned fixations.", 

541 code="scanpath_studio/measures.py:compute_per_word_measures", 

542 output="n_fixations", 

543 missing="Never fixated ⇒ 0.", 

544 tiers="A", 

545 status=STATUS_VERIFIED, 

546 consumers=(_UI, _API), 

547 tests=("tests/test_synthetic.py",), 

548 ), 

549 Computation( 

550 id="measure.skip", 

551 name="Skip flag / skip rate", 

552 category=CATEGORY_MEASURE, 

553 summary="Whether a word received no first-pass fixation.", 

554 formula="`skip_flag = no fixation in the word's first pass`.", 

555 code="scanpath_studio/measures.py:compute_per_word_measures", 

556 output="skip_flag", 

557 unit="rate when aggregated (0–1)", 

558 missing="A word fixated only after a regression still counts as skipped.", 

559 tiers="A", 

560 status=STATUS_VERIFIED, 

561 consumers=(_UI, _API), 

562 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

563 ), 

564 Computation( 

565 id="measure.regressions", 

566 name="Regression in/out flags", 

567 category=CATEGORY_MEASURE, 

568 summary="Whether a word was returned to, or left backwards.", 

569 formula=( 

570 "`regression_in_flag` — some later fixation lands on this word after " 

571 "the gaze had moved past it. `regression_out_flag` — a regression " 

572 "to an earlier word is made from this word during first pass, before " 

573 "the eyes first leave it forwards (EyeLink's `IA_REGRESSION_OUT`, " 

574 "BUG-64); a regression from it later in the trial does not count." 

575 ), 

576 code="scanpath_studio/measures.py:compute_per_word_measures", 

577 output="regression_in_flag, regression_out_flag", 

578 unit="rate when aggregated (0–1)", 

579 precedence="Precomputed IA regression flags win (see `norm.flags`).", 

580 tiers="A", 

581 status=STATUS_PARTIAL, 

582 consumers=(_UI, _API), 

583 tests=("tests/test_measures.py", "tests/test_synthetic.py"), 

584 ), 

585 Computation( 

586 id="measure.landing_position", 

587 name="Initial landing position", 

588 category=CATEGORY_MEASURE, 

589 summary="Where in the word the first fixation landed, in letters.", 

590 formula=( 

591 "`char_width = geom.word_char_advance`; " 

592 "`offset = first_fix_x − word.x` (LTR) or " 

593 "`word.x + n·advance − first_fix_x` (RTL, BUG-27); " 

594 "`landing_position = offset / char_width + 1` — so the first letter " 

595 "starts at 1 and its center is 1.5. Unclipped: on a tiling corpus " 

596 "the box's last cell is the space after the word, which belongs to " 

597 "it (#BUG-83), so a first fixation there reads `n + 1` to `n + 2`." 

598 ), 

599 code="scanpath_studio/measures.py:compute_per_word_measures", 

600 output="initial_landing_position", 

601 unit="letters", 

602 missing=( 

603 "Never fixated, zero width, or no text ⇒ NaN. Measured from the " 

604 "word's first fixation, first pass or not (as `measure.ffd`)." 

605 ), 

606 precedence=( 

607 "VAL-5: the scale is `geom.word_char_advance`, not " 

608 "`width / len(text)`, which on a tiling corpus puts every landing " 

609 "~`(n+1)/n` too far into the word." 

610 ), 

611 tiers="A", 

612 status=STATUS_PARTIAL, 

613 reference=( 

614 "Assumes a monospaced advance within the word box — exact for the " 

615 "app's monospace default, approximate for proportional fonts." 

616 ), 

617 consumers=(_UI, _API), 

618 tests=("tests/test_measures.py",), 

619 ), 

620 Computation( 

621 id="measure.landing_distance", 

622 name="Centred landing distance", 

623 category=CATEGORY_MEASURE, 

624 summary="Landing position relative to the word's center.", 

625 formula=( 

626 "`landing_position − (1 + len(text) / 2)` — the glyphs span " 

627 "`[1, n + 1)`, so that is the word's center (BUG-65). The center of " 

628 "the *letters*, not of the box: a tiling box's trailing space " 

629 "(#BUG-83) would move it half a letter right." 

630 ), 

631 code="scanpath_studio/measures.py:compute_per_word_measures", 

632 output="initial_landing_distance", 

633 unit="letters (0 = word center, negative = left of center)", 

634 missing="As `measure.landing_position`.", 

635 tiers="A", 

636 status=STATUS_PARTIAL, 

637 consumers=(_UI, _API), 

638 tests=("tests/test_measures.py",), 

639 ), 

640 Computation( 

641 id="measure.second_pass", 

642 name="Second-pass duration", 

643 category=CATEGORY_MEASURE, 

644 summary="Time spent on the word during its second visit.", 

645 formula="Sum of the fixations in the word's second run.", 

646 code="scanpath_studio/measures.py:compute_per_word_measures", 

647 output="second_pass_duration_ms", 

648 unit="ms", 

649 missing=( 

650 "Fewer than two runs ⇒ 0. An imported blank " 

651 "`IA_SECOND_RUN_DWELL_TIME` becomes 0 too, where the fixation count " 

652 "is known." 

653 ), 

654 tiers="A", 

655 status=STATUS_PARTIAL, 

656 consumers=(_UI, _API), 

657 tests=("tests/test_measures.py",), 

658 ), 

659 Computation( 

660 id="measure.single_fix", 

661 name="Single-fixation duration", 

662 category=CATEGORY_MEASURE, 

663 summary="First-pass duration when the first pass was exactly one fixation.", 

664 formula="FFD when the word's first run has length 1, else NaN.", 

665 code="scanpath_studio/measures.py:compute_per_word_measures", 

666 output="single_fixation_duration_ms", 

667 unit="ms", 

668 missing=("A first run of more than one fixation ⇒ NaN. " + _UNFIXATED_MISSING), 

669 tiers="A", 

670 status=STATUS_PARTIAL, 

671 reference="Rayner (1998).", 

672 consumers=(_UI, _API), 

673 tests=("tests/test_measures.py",), 

674 ), 

675 Computation( 

676 id="measure.reg_in_count", 

677 name="Regressions into word", 

678 category=CATEGORY_MEASURE, 

679 summary="How many times the gaze came back to this word.", 

680 formula=( 

681 "Number of regressions into the word — entries from a later word " 

682 "(EyeLink's `IA_REGRESSION_IN_COUNT`). A re-entry from an *earlier* " 

683 "word is a new run but not a regression in." 

684 ), 

685 code="scanpath_studio/measures.py:compute_per_word_measures", 

686 output="number_of_regressions_in", 

687 missing="Never regressed into ⇒ 0.", 

688 tiers="A", 

689 status=STATUS_PARTIAL, 

690 consumers=(_UI, _API), 

691 tests=("tests/test_measures.py",), 

692 ), 

693 Computation( 

694 id="fix.saccade_amplitude", 

695 name="Saccade amplitude", 

696 category=CATEGORY_MEASURE, 

697 summary="Distance between consecutive fixations — always pixels (BUG-25).", 

698 formula="`sqrt(dx² + dy²)` between consecutive fixations in the trial.", 

699 code="scanpath_studio/measures.py:enrich_fixations", 

700 output="saccade_amplitude", 

701 unit="px", 

702 grouping="Per trial, in timestamp order; the first fixation has none.", 

703 missing="First fixation of a trial ⇒ NaN.", 

704 precedence=( 

705 "A source column literally named `saccade_amplitude` is assumed to " 

706 "be pixels and kept. EyeLink's **degree**-valued " 

707 "`NEXT_SAC_AMPLITUDE` / `PREVIOUS_SAC_AMPLITUDE` normalize to " 

708 "`next_/prev_saccade_amplitude_deg` and never reach this column — " 

709 "they are different quantities *and* different saccades." 

710 ), 

711 tiers="A, C", 

712 status=STATUS_VERIFIED, 

713 # The history (one column once meant px or deg) is in the changelog. 

714 reference="(BUG-25)", 

715 consumers=(_UI, _API, _EXPORT, _CORPUS), 

716 tests=("tests/test_measures.py",), 

717 ), 

718 Computation( 

719 id="fix.angles", 

720 name="Saccade angles", 

721 category=CATEGORY_MEASURE, 

722 summary="Incoming and outgoing saccade direction.", 

723 formula=( 

724 "`angle_incoming = degrees(atan2(−dy, dx))` from the previous " 

725 "fixation; `angle_outgoing` is the next fixation's incoming angle. " 

726 "`−dy` because screen y grows downwards, so 0° is rightward and " 

727 "positive is up." 

728 ), 

729 code="scanpath_studio/measures.py:enrich_fixations", 

730 output="angle_incoming, angle_outgoing", 

731 unit="degrees (−180, 180]", 

732 missing="Trial edges ⇒ NaN.", 

733 tiers="A, C", 

734 status=STATUS_VERIFIED, 

735 consumers=(_UI, _API, _EXPORT), 

736 tests=("tests/test_measures.py",), 

737 ), 

738 Computation( 

739 id="fix.rebased_onsets", 

740 name="Rebased fixation onsets", 

741 category=CATEGORY_MEASURE, 

742 summary="Trial-relative onset times for animation and time series.", 

743 formula=( 

744 "Cumulative onsets rebased so the trial starts at 0, from " 

745 "`timestamp_ms` where present, else by accumulating durations." 

746 ), 

747 code="scanpath_studio/measures.py:rebased_fixation_onsets", 

748 output="Onset array", 

749 unit="ms", 

750 missing="A backwards clock restarts the accumulation (see VAL-7).", 

751 tiers="A, C", 

752 status=STATUS_PARTIAL, 

753 consumers=(_UI, _API, _CLI), 

754 tests=("tests/test_measures.py",), 

755 ), 

756 # ------------------------------------------------------------------ 

757 # Preprocessing 

758 # ------------------------------------------------------------------ 

759 Computation( 

760 id="pre.merge_short", 

761 name="Short-fixation merging", 

762 category=CATEGORY_PREPROCESSING, 

763 summary="Fold a short fixation into a neighbour within a character distance.", 

764 formula=( 

765 "A fixation below the short threshold is merged into the nearer " 

766 "adjacent fixation when that neighbour is within the merge distance, " 

767 "expressed in characters and converted to px via " 

768 "`geom.word_char_advance`. Durations add; position follows the " 

769 "survivor." 

770 ), 

771 precedence=( 

772 "#BUG-27: the conversion reads the shared letter scale, so " 

773 '"within 1 character" means the same on every word.' 

774 ), 

775 code="scanpath_studio/preprocessing.py:merge_short_fixations", 

776 output="A reduced fixation frame", 

777 unit="ms threshold, characters distance", 

778 missing="Off by default; original rows stay available.", 

779 tiers="A, C", 

780 status=STATUS_PARTIAL, 

781 reference="A common cleaning step; thresholds are the user's choice.", 

782 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT), 

783 tests=("tests/test_preprocessing.py",), 

784 ), 

785 Computation( 

786 id="pre.exclude_short", 

787 name="Short/long fixation exclusion", 

788 category=CATEGORY_PREPROCESSING, 

789 summary="Soft-exclude fixations outside a duration window.", 

790 formula="Drop fixations shorter than / longer than the chosen bounds.", 

791 code="scanpath_studio/preprocessing.py:preprocess_fixations", 

792 unit="ms", 

793 missing="Soft: excluded rows are reported, not deleted from the source.", 

794 tiers="C", 

795 status=STATUS_PARTIAL, 

796 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT), 

797 tests=("tests/test_preprocessing.py",), 

798 ), 

799 Computation( 

800 id="pre.blink_adjacent", 

801 name="Blink-adjacent exclusion", 

802 category=CATEGORY_PREPROCESSING, 

803 summary="Drop fixations immediately before/after a blink.", 

804 formula="Exclude the fixations neighbouring any row flagged `is_blink`.", 

805 code="scanpath_studio/preprocessing.py:preprocess_fixations", 

806 missing="No blink column ⇒ the option has no effect.", 

807 tiers="C", 

808 status=STATUS_PARTIAL, 

809 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT), 

810 tests=("tests/test_preprocessing.py",), 

811 ), 

812 Computation( 

813 id="pre.cleaning_report", 

814 name="Cleaning QA report", 

815 category=CATEGORY_PREPROCESSING, 

816 summary="What the preprocessing pass would remove, and why.", 

817 formula="Counts per exclusion reason over the unfiltered frame.", 

818 code="scanpath_studio/preprocessing.py:cleaning_report", 

819 output="Cleaning QA table", 

820 tiers="C", 

821 status=STATUS_PARTIAL, 

822 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT, _INSPECT_DERIVED), 

823 tests=("tests/test_preprocessing.py",), 

824 ), 

825 Computation( 

826 id="pre.sentence_measures", 

827 name="Sentence-level measures", 

828 category=CATEGORY_PREPROCESSING, 

829 summary="Per-sentence reading time and counts.", 

830 formula=( 

831 "Words are grouped into sentences by `infer_sentence_ids` " 

832 "(terminal punctuation). Each sentence's durations, fixation and " 

833 "run counts, go-past times and skip flag are then derived from the " 

834 "fixations on its words; the supplied word measures are not used, " 

835 "so a sentence with no fixations reads as skipped (which is why " 

836 "Corpus Analysis → Per sentence is held back)." 

837 ), 

838 code="scanpath_studio/preprocessing.py:sentence_measures", 

839 output="Sentences table", 

840 unit="ms, counts", 

841 missing="Sentence inference is textual, not annotated — approximate.", 

842 tiers="C", 

843 status=STATUS_PARTIAL, 

844 consumers=(_CORPUS, _API, _CLI, _EXPORT, _INSPECT_DERIVED), 

845 tests=("tests/test_preprocessing.py",), 

846 ), 

847 Computation( 

848 id="pre.saccade_table", 

849 name="Saccade table", 

850 category=CATEGORY_PREPROCESSING, 

851 summary="One row per saccade, with amplitude, angle and class.", 

852 formula=( 

853 "Consecutive fixation pairs within a trial; amplitude in px, and in " 

854 "degrees only when `pixels_per_degree` is supplied." 

855 ), 

856 code="scanpath_studio/preprocessing.py:saccade_table", 

857 output="Saccades table", 

858 unit="px, deg (when geometry is known), ms", 

859 missing="Assumed geometry ⇒ the degree columns inherit that assumption.", 

860 tiers="C", 

861 status=STATUS_PARTIAL, 

862 consumers=(_API, _CLI, _EXPORT, _INSPECT_DERIVED), 

863 tests=("tests/test_preprocessing.py",), 

864 ), 

865 Computation( 

866 id="pre.character_grid", 

867 name="Character grid", 

868 category=CATEGORY_PREPROCESSING, 

869 summary="Per-character boxes derived from word boxes.", 

870 formula=( 

871 "Character `k` of a word spans `x + (k−1) × advance` to `x + k × " 

872 "advance`, where the advance is `geom.word_char_advance`." 

873 ), 

874 code="scanpath_studio/preprocessing.py:character_grid", 

875 unit="px", 

876 missing="Proportional fonts make this an approximation.", 

877 precedence=( 

878 "#BUG-27: the advance is the shared letter scale, not `width / " 

879 "len(text)` — which on a tiling corpus stretched the glyph row " 

880 "across the trailing inter-word padding, so each character box after " 

881 "the first sat progressively further right than its glyph." 

882 ), 

883 tiers="A, C", 

884 status=STATUS_CONVENTION, 

885 consumers=(_API, _CLI, _EXPORT, _INSPECT_DERIVED), 

886 tests=("tests/test_preprocessing.py",), 

887 ), 

888 Computation( 

889 id="pre.rtl", 

890 name="Right-to-left detection", 

891 category=CATEGORY_PREPROCESSING, 

892 summary="Whether a word's script runs right to left.", 

893 formula="Unicode range test over the word's characters.", 

894 code="scanpath_studio/preprocessing.py:detect_right_to_left", 

895 output="right_to_left", 

896 tiers="A, C", 

897 status=STATUS_VERIFIED, 

898 consumers=(_UI, _API, _CORPUS), 

899 tests=("tests/test_preprocessing.py",), 

900 ), 

901 Computation( 

902 id="pre.sensitivity", 

903 name="Measure sensitivity", 

904 category=CATEGORY_PREPROCESSING, 

905 summary="How much the word measures move under different line assignments.", 

906 formula=( 

907 "Each trial's fixations are line-assigned by every method in " 

908 "`methods` (default `attach`, `slice`, `consensus`), FFD / FPRT / " 

909 "RPD / TFD are recomputed per method, and each word's spread (max − " 

910 "min across methods) is reported beside a per-trial correction " 

911 "report (PRE-18)." 

912 ), 

913 code="scanpath_studio/preprocessing.py:measure_sensitivity", 

914 tiers="C", 

915 status=STATUS_PARTIAL, 

916 consumers=(_API_EXPERIMENTAL,), 

917 tests=("tests/test_preprocessing.py",), 

918 ), 

919 Computation( 

920 id="align.algorithms", 

921 name="Vertical drift correction", 

922 category=CATEGORY_PREPROCESSING, 

923 summary="Line-assignment algorithms, ported natively (PRE-3).", 

924 formula=( 

925 "The ten Carr et al. algorithms — `attach`, `chain`, `cluster`, " 

926 "`compare`, `merge`, `regress`, `segment`, `split`, `stretch`, " 

927 "`warp` — plus `slice` and a `consensus` vote over them. Each " 

928 "reassigns fixation *y* to a text line. Not in this release " 

929 "(#PRE-21)." 

930 ), 

931 code="scanpath_studio/alignment.py:correct", 

932 output="Corrected fixation y (display only; exported tables stay raw)", 

933 missing="Off by default; the original coordinates are never overwritten.", 

934 tiers="B, C", 

935 status=STATUS_PARTIAL, 

936 reference=( 

937 "Carr, Pescuma, Furlan, Ktori & Crepaldi (2021), *Algorithms for the " 

938 "automated correction of vertical drift in eye-tracking data*, " 

939 "Behavior Research Methods. Ported from the reference implementation " 

940 "— the one entry with a genuine tier-B comparison." 

941 ), 

942 consumers=(_UI, _API, _CLI), 

943 tests=("tests/test_alignment.py", "tests/test_cli_drift.py"), 

944 ), 

945 # ------------------------------------------------------------------ 

946 # Aggregation / statistics 

947 # ------------------------------------------------------------------ 

948 Computation( 

949 id="agg.measure_values", 

950 name="Measure value extraction", 

951 category=CATEGORY_AGGREGATION, 

952 summary="Pull one registered measure's values out of a frame.", 

953 formula=( 

954 "The `aggregation.MEASURES` entry names the frame (words or " 

955 "fixations), the column and the unit; values are coerced numeric and " 

956 "NaNs dropped." 

957 ), 

958 code="scanpath_studio/aggregation.py:measure_values", 

959 missing="Non-numeric entries become NaN and are dropped, not zeroed.", 

960 tiers="C, D", 

961 status=STATUS_PARTIAL, 

962 consumers=(_CORPUS, _API), 

963 tests=("tests/test_aggregation.py",), 

964 ), 

965 Computation( 

966 id="agg.aggregate_value", 

967 name="Central tendency", 

968 category=CATEGORY_AGGREGATION, 

969 summary="The Aggregate selector: mean / median / sum.", 

970 formula="`np.nanmean` · `np.nanmedian` · `np.nansum` over the values.", 

971 code="scanpath_studio/aggregation.py:aggregate_value", 

972 missing="NaN-skipping throughout; an all-NaN input gives NaN.", 

973 tiers="A, C", 

974 status=STATUS_VERIFIED, 

975 consumers=(_CORPUS, _API), 

976 tests=("tests/test_aggregation.py",), 

977 ), 

978 Computation( 

979 id="agg.spread", 

980 name="Spread band", 

981 category=CATEGORY_AGGREGATION, 

982 summary="The error band drawn around an aggregate.", 

983 formula=( 

984 "`SD` → ±1 sample std (ddof=1) · `SEM` → ±std/√n · `IQR` → the 25th " 

985 "and 75th percentiles · `Bootstrap CI` → `agg.bootstrap_ci`. With " 

986 "`agg='sum'`, SD/SEM fall back to the bootstrap: the spread of " 

987 "individual observations does not bracket a total." 

988 ), 

989 code="scanpath_studio/aggregation.py:spread_bounds", 

990 missing="Empty input or NaN center ⇒ a zero-width band.", 

991 tiers="A, C", 

992 status=STATUS_VERIFIED, 

993 consumers=(_CORPUS, _API), 

994 tests=("tests/test_aggregation.py",), 

995 ), 

996 Computation( 

997 id="agg.bootstrap_ci", 

998 name="Bootstrap confidence interval", 

999 category=CATEGORY_AGGREGATION, 

1000 summary="Percentile bootstrap CI of the chosen aggregate.", 

1001 formula=( 

1002 "1000 resamples with replacement; the CI is the 2.5th and 97.5th " 

1003 "percentiles of the resampled statistic." 

1004 ), 

1005 code="scanpath_studio/aggregation.py:bootstrap_ci", 

1006 unit="same as the measure", 

1007 missing="n < 2 ⇒ a degenerate interval at the point estimate.", 

1008 precedence="Seeded (`seed=0`) — the same data gives the same interval.", 

1009 tiers="A, C", 

1010 status=STATUS_VERIFIED, 

1011 reference="Percentile bootstrap; no bias correction.", 

1012 consumers=(_CORPUS, _API), 

1013 tests=("tests/test_aggregation.py",), 

1014 ), 

1015 Computation( 

1016 id="agg.effect_size", 

1017 name="Group means and difference", 

1018 category=CATEGORY_AGGREGATION, 

1019 summary="Two groups' means, their difference and Cohen's d (AN-21).", 

1020 formula=( 

1021 "Each value is one participant's mean of the measure (pooled " 

1022 "observations when the data names no participants). " 

1023 "`mean_diff = mean(A) − mean(B)`. Cohen's *d* uses the pooled SD " 

1024 "`sqrt(((nA−1)·varA + (nB−1)·varB) / (nA+nB−2))` with ddof=1, and " 

1025 "is shown only when the groups share no participant." 

1026 ), 

1027 code="scanpath_studio/aggregation.py:group_mean_difference", 

1028 output="mean_a, mean_b, mean_diff, cohen_d, n_a, n_b", 

1029 grouping="One value per participant in each group", 

1030 missing=( 

1031 "n < 2 in either group ⇒ NaN *d*. A zero pooled SD gives " 

1032 "**NaN**, not 0.0, so it cannot read as 'no effect' beside a " 

1033 "non-zero mean difference." 

1034 ), 

1035 tiers="A, C", 

1036 status=STATUS_PARTIAL, 

1037 reference=( 

1038 "**Descriptive only** — no significance test. A participant in both " 

1039 "groups contributes to both means, so the groups are not " 

1040 "independent samples." 

1041 ), 

1042 consumers=(_CORPUS, _API), 

1043 tests=("tests/test_aggregation.py",), 

1044 ), 

1045 Computation( 

1046 id="agg.group_mask", 

1047 name="Group definition", 

1048 category=CATEGORY_AGGREGATION, 

1049 summary="Which rows belong to a cohort.", 

1050 formula=( 

1051 "A spec maps column → allowed values; the mask is the conjunction of " 

1052 "membership tests. Two modes: split one field, or two independent " 

1053 "filter sets. A key may be a tuple of columns matched as one " 

1054 "composite key: a trial-metadata field resolves to the " 

1055 "(participant, trial) readings its rows describe, a participant field to " 

1056 "participant ids and a text field to text ids — the tables are never " 

1057 "joined onto the frames." 

1058 ), 

1059 code="scanpath_studio/aggregation.py:group_mask", 

1060 missing=( 

1061 "A column (or any column of a composite key) absent from the frame " 

1062 "contributes no constraint; a metadata selection that matches " 

1063 "nothing selects no rows." 

1064 ), 

1065 tiers="A, C", 

1066 status=STATUS_VERIFIED, 

1067 consumers=(_CORPUS, _API), 

1068 tests=("tests/test_aggregation.py",), 

1069 ), 

1070 Computation( 

1071 id="agg.word_profile", 

1072 name="Per-word cohort profile", 

1073 category=CATEGORY_AGGREGATION, 

1074 summary="A measure per word position, aggregated across participants.", 

1075 formula="Group the word measures by word id and apply `agg.aggregate_value`.", 

1076 code="scanpath_studio/aggregation.py:cohort_word_profile", 

1077 missing="A minimum-participants threshold drops thinly-sampled words.", 

1078 tiers="C", 

1079 status=STATUS_PARTIAL, 

1080 consumers=(_CORPUS, _API), 

1081 tests=("tests/test_aggregation.py",), 

1082 ), 

1083 Computation( 

1084 id="agg.word_rates", 

1085 name="Skip / regression rate profile", 

1086 category=CATEGORY_AGGREGATION, 

1087 summary="Rate measures per word.", 

1088 formula=( 

1089 "Mean of the 0/1 flag over the participants who reported it — a " 

1090 "proportion in [0, 1]. Each rate has its own participant count " 

1091 "(`n_skip`, `n_regression_in`) and its own minimum-participants verdict." 

1092 ), 

1093 code="scanpath_studio/aggregation.py:word_rate_profile", 

1094 unit="proportion", 

1095 missing=( 

1096 "A missing flag is no observation: it is left out of that rate and " 

1097 "its participant count, never read as 0. A rate below the minimum " 

1098 "participants is hidden; the word stays while its other rate stands." 

1099 ), 

1100 tiers="A, C", 

1101 status=STATUS_PARTIAL, 

1102 consumers=(_CORPUS, _API), 

1103 tests=("tests/test_aggregation.py",), 

1104 ), 

1105 Computation( 

1106 id="agg.reader_summary", 

1107 name="Per-participant summary", 

1108 category=CATEGORY_AGGREGATION, 

1109 summary="One row per participant: totals, means and rates.", 

1110 formula=( 

1111 "Counts and NaN-skipping means over that participant's rows. " 

1112 "`mean_saccade_px` is the mean of `fix.saccade_amplitude` and is " 

1113 "in pixels." 

1114 ), 

1115 code="scanpath_studio/aggregation.py:reader_summary_table", 

1116 output="Readers table", 

1117 unit="ms, px, counts, proportions", 

1118 tiers="C, D", 

1119 status=STATUS_PARTIAL, 

1120 consumers=(_CORPUS, _EXPORT, _INSPECT, _API), 

1121 tests=("tests/test_aggregation.py",), 

1122 ), 

1123 Computation( 

1124 id="agg.trial_summary", 

1125 name="Per-trial summary", 

1126 category=CATEGORY_AGGREGATION, 

1127 summary="One row per trial: reading time, counts, rates.", 

1128 formula=( 

1129 "Counts and sums over the trial's fixations and word measures. " 

1130 "`reading_time_ms` is last fixation end − first fixation start; " 

1131 "without recorded fixation onsets it is the summed fixation " 

1132 "durations, and `reading_time_source` says it is an estimate. " 

1133 "`wpm` = words ÷ reading time." 

1134 ), 

1135 missing=( 

1136 "No onset column ⇒ reading time and wpm are duration-based " 

1137 "estimates, labeled as such — never the 0, 1, 2, … order numbers." 

1138 ), 

1139 code="scanpath_studio/aggregation.py:trial_summary_table", 

1140 output="Trials table", 

1141 unit="ms, counts", 

1142 tiers="C, D", 

1143 status=STATUS_PARTIAL, 

1144 consumers=(_CORPUS, _EXPORT, _INSPECT, _API), 

1145 tests=("tests/test_aggregation.py",), 

1146 ), 

1147 Computation( 

1148 id="agg.normalize", 

1149 name="Normalized measure column", 

1150 category=CATEGORY_AGGREGATION, 

1151 summary="Rescale a measure for cross-participant comparison.", 

1152 formula=( 

1153 "Per-participant z-score, `(value − participant mean) / participant SD`, " 

1154 "when **Z-score per participant** is on." 

1155 ), 

1156 code="scanpath_studio/aggregation.py:add_normalized_column", 

1157 missing=( 

1158 "A participant with zero variance (or one value) ⇒ 0, the participant's own " 

1159 "mean; a missing value stays NaN." 

1160 ), 

1161 tiers="A, C", 

1162 status=STATUS_PARTIAL, 

1163 consumers=(_CORPUS,), 

1164 tests=("tests/test_aggregation.py",), 

1165 ), 

1166 Computation( 

1167 id="agg.landing_curve", 

1168 name="Landing-position curve", 

1169 category=CATEGORY_AGGREGATION, 

1170 summary="Distribution of initial landing positions by word length.", 

1171 formula=( 

1172 "Histogram of the landing position as a *fraction of the word's " 

1173 "interest area* — `(first_fix_x − word.x) / width` over the " 

1174 "experiment's own box, i.e. `(measure.landing_position − 1)` over " 

1175 "the box's `width / geom.word_char_advance` character cells (RTL " 

1176 "counted from where the glyphs end, as the letter position is). " 

1177 "Unclipped — binned per word length." 

1178 ), 

1179 code="scanpath_studio/aggregation.py:landing_positions", 

1180 unit=( 

1181 "fraction of the interest area (0–1 for a landing inside the box), " 

1182 "or px with `as_fraction=False`" 

1183 ), 

1184 precedence=( 

1185 "#BUG-83: on a glyph-tight corpus the box is the glyph run, so 0 is " 

1186 "the first letter's edge and 1 the last's. On a tiling corpus the " 

1187 "box's last cell is the space after the word, so the glyphs fill " 

1188 "`[0, n / (n + 1))` and a landing on that space reads just below 1, " 

1189 "not clipped onto 1.0. A first fixation assigned from outside the box " 

1190 "(by an imported `word_id`) reads below 0 or above 1 rather than " 

1191 "being clipped onto an edge. The origin is the word's `x` and the " 

1192 "scale is `geom.word_char_advance`." 

1193 ), 

1194 tiers="C", 

1195 status=STATUS_PARTIAL, 

1196 consumers=(_CORPUS,), 

1197 tests=("tests/test_aggregation.py",), 

1198 ), 

1199 Computation( 

1200 id="agg.over_time", 

1201 name="Trend over time", 

1202 category=CATEGORY_AGGREGATION, 

1203 summary="A measure by trial index or fixation index.", 

1204 formula="Aggregate per index position across the selection.", 

1205 code="scanpath_studio/aggregation.py:metric_over_time", 

1206 missing="Index positions with no data are gaps, not zeros.", 

1207 tiers="C", 

1208 status=STATUS_PARTIAL, 

1209 consumers=(_CORPUS,), 

1210 tests=("tests/test_aggregation.py",), 

1211 ), 

1212 # ------------------------------------------------------------------ 

1213 # Similarity 

1214 # ------------------------------------------------------------------ 

1215 Computation( 

1216 id="sim.nld", 

1217 name="Normalized Levenshtein distance", 

1218 category=CATEGORY_SIMILARITY, 

1219 summary="Scanpath similarity over AoI sequences.", 

1220 formula=( 

1221 "`levenshtein(a, b) / max(len(a), len(b))` ∈ [0, 1]; 0 is identical. " 

1222 "Two empty sequences give 0." 

1223 ), 

1224 code="scanpath_studio/similarity.py:normalized_levenshtein", 

1225 unit="dimensionless (0–1)", 

1226 missing="Not in this release.", 

1227 tiers="A, C", 

1228 status=STATUS_VERIFIED, 

1229 reference="Standard edit-distance scanpath comparison.", 

1230 consumers=(_UI, _API), 

1231 tests=("tests/test_similarity.py",), 

1232 ), 

1233 Computation( 

1234 id="sim.aoi_sequence", 

1235 name="AoI sequence", 

1236 category=CATEGORY_SIMILARITY, 

1237 summary="The symbol string an NLD comparison runs on.", 

1238 formula=( 

1239 "Assigned `word_id`s in fixation order, with unassigned fixations " 

1240 "dropped and (optionally) immediate repeats collapsed." 

1241 ), 

1242 code="scanpath_studio/similarity.py:aoi_sequence", 

1243 missing="A trial with no assigned fixations yields an empty sequence.", 

1244 tiers="A, C", 

1245 status=STATUS_VERIFIED, 

1246 consumers=(_UI, _API), 

1247 tests=("tests/test_similarity.py",), 

1248 ), 

1249 Computation( 

1250 id="sim.windowed", 

1251 name="NLD by fixation index / time", 

1252 category=CATEGORY_SIMILARITY, 

1253 summary="Similarity restricted to a window of the scanpath.", 

1254 formula="`sim.nld` over the sub-sequence inside the index or time window.", 

1255 code="scanpath_studio/similarity.py:nld_by_fixation_index", 

1256 tiers="C", 

1257 status=STATUS_PARTIAL, 

1258 consumers=(_UI, _API), 

1259 tests=("tests/test_similarity.py",), 

1260 ), 

1261 # ------------------------------------------------------------------ 

1262 # Unit / coordinate conversion 

1263 # ------------------------------------------------------------------ 

1264 Computation( 

1265 id="geom.pixels_per_degree", 

1266 name="Pixels per degree of visual angle", 

1267 category=CATEGORY_GEOMETRY, 

1268 summary="The screen-geometry conversion every angular unit depends on.", 

1269 formula=( 

1270 "`px_per_mm = canvas_width_px / monitor_width_mm`; " 

1271 "`mm_per_degree = 2 · viewing_distance_mm · tan(0.5°)`; " 

1272 "`px_per_degree = px_per_mm · mm_per_degree`." 

1273 ), 

1274 code="scanpath_studio/experimental_setup.py:pixels_per_degree", 

1275 unit="px / degree", 

1276 missing="Any missing geometry ⇒ no conversion is offered at all.", 

1277 precedence=( 

1278 "**Provenance matters more than the number.** Every built-in corpus " 

1279 "assumes its monitor size and viewing distance, so a degree-valued " 

1280 "result inherits that — see the *Recording setup* panel." 

1281 ), 

1282 tiers="A, C", 

1283 status=STATUS_VERIFIED, 

1284 consumers=(_UI, _API, _CLI, _EXPORT), 

1285 tests=("tests/test_experimental_setup.py",), 

1286 ), 

1287 Computation( 

1288 id="geom.font_pt_to_px", 

1289 name="Font point size to pixels", 

1290 category=CATEGORY_GEOMETRY, 

1291 summary="Typography conversion for true-scale text rendering.", 

1292 formula="`px = pt · dpi / 72`.", 

1293 code="scanpath_studio/experimental_setup.py:font_pt_to_px", 

1294 unit="px", 

1295 tiers="A, C", 

1296 status=STATUS_VERIFIED, 

1297 consumers=(_UI, _API, _CLI), 

1298 tests=("tests/test_experimental_setup.py",), 

1299 ), 

1300 Computation( 

1301 id="geom.word_box_bounds", 

1302 name="Word interest-area edges", 

1303 category=CATEGORY_GEOMETRY, 

1304 summary="Where one word's interest area ends and the next begins.", 

1305 formula=( 

1306 "`x .. x + width` by `y .. y + height` — the experiment's own " 

1307 "rectangles, unmodified. On a tiling corpus each box includes the " 

1308 "space after its word." 

1309 ), 

1310 code="scanpath_studio/measures.py:word_box_bounds", 

1311 unit="px", 

1312 precedence=( 

1313 "The boundary *between* words, for everything that tests a point " 

1314 "against a box or draws one: `assign.fixation_to_word`, " 

1315 "`assign.in_text`, the drawn outlines, the word heatmaps, the " 

1316 "critical-span frame, drift correction and the model scanpaths. A " 

1317 "position *inside* a word goes through `geom.word_char_advance` " 

1318 "instead, and where its letters are through `geom.word_glyph_span`; the drawn word label is centered in the box (#BUG-97)." 

1319 ), 

1320 tiers="A, C", 

1321 status=STATUS_PARTIAL, 

1322 consumers=(_UI, _API, _CORPUS), 

1323 tests=("tests/test_word_box_geometry.py", "tests/test_word_id_offset.py"), 

1324 ), 

1325 Computation( 

1326 id="geom.word_box_space_px", 

1327 name="Inter-word padding baked into each box", 

1328 category=CATEGORY_GEOMETRY, 

1329 summary="Detects a tiling layout that carries one trailing space per box.", 

1330 formula=( 

1331 "Median of `width / (len(text) + 1)` across one trial's words — the " 

1332 "advance — reported only when the boxes are consistently that wide " 

1333 "**and** actually tile (no gaps). Anything else ⇒ `0.0`, i.e. " 

1334 "'these AOIs are glyph-tight — each box is its glyph run'." 

1335 ), 

1336 code="scanpath_studio/measures.py:word_box_space_px", 

1337 unit="px", 

1338 missing="No usable words ⇒ 0.0 (glyph-tight), never a guess.", 

1339 precedence=( 

1340 "Never moves a box edge (#BUG-83); it only tells " 

1341 "`geom.word_char_advance` and `geom.word_glyph_span` how many " 

1342 "character cells a box holds." 

1343 ), 

1344 tiers="A, C", 

1345 status=STATUS_VERIFIED, 

1346 consumers=(_UI, _API, _EXPORT), 

1347 tests=("tests/test_measures.py",), 

1348 ), 

1349 Computation( 

1350 id="geom.word_char_advance", 

1351 name="Character advance within a word", 

1352 category=CATEGORY_GEOMETRY, 

1353 summary="How wide one letter is — the scale for every within-word position.", 

1354 formula=( 

1355 "`width / (len(text) + 1)` when `geom.word_box_space_px` finds " 

1356 "trailing padding, else `width / len(text)`." 

1357 ), 

1358 code="scanpath_studio/measures.py:word_char_advance", 

1359 unit="px / character", 

1360 missing="No `text`/`width` ⇒ NaN, and the letter measures report NaN.", 

1361 precedence=( 

1362 "The single accessor for the letter scale, as `geom.word_box_bounds` " 

1363 "is for the boundary between words: `measure.landing_position`, " 

1364 "`measure.landing_distance`, `agg.landing_curve` and the saccade " 

1365 "table's launch/landing letter all read it. #BUG-27 — before that " 

1366 "each derived its own `width / len(text)`, which is one advance too " 

1367 "wide on a tiling corpus, by a factor that varied with word length." 

1368 ), 

1369 tiers="A, C", 

1370 status=STATUS_VERIFIED, 

1371 consumers=(_UI, _API, _EXPORT, _CORPUS), 

1372 tests=("tests/test_measures.py",), 

1373 ), 

1374 Computation( 

1375 id="geom.word_glyph_span", 

1376 name="Where a word's glyphs are", 

1377 category=CATEGORY_GEOMETRY, 

1378 summary="The glyph run inside a word's box — where its letters are.", 

1379 formula=( 

1380 "Starts at `x` and runs `len(text) × geom.word_char_advance`: the " 

1381 "whole box on a glyph-tight corpus, one advance short of it on a " 

1382 "tiling one. No `text` ⇒ the box width." 

1383 ), 

1384 code="scanpath_studio/measures.py:word_glyph_span", 

1385 unit="px", 

1386 precedence=( 

1387 "Not an interest area: `agg.landing_curve` measures a landing " 

1388 "across it and mirrors an RTL one. Measured against " 

1389 "OneStop's own Experiment Builder screens: each tiling box is " 

1390 "centered on its word, half a space either side, so the run's " 

1391 "`x` start is half an advance early there; the label and snap " 

1392 "use the box center, the landing measures do not." 

1393 ), 

1394 tiers="A", 

1395 status=STATUS_PARTIAL, 

1396 consumers=(_UI, _API, _CORPUS), 

1397 tests=("tests/test_word_box_geometry.py",), 

1398 ), 

1399 # ------------------------------------------------------------------ 

1400 # Display / export transformations 

1401 # ------------------------------------------------------------------ 

1402 Computation( 

1403 id="disp.marker_sizes", 

1404 name="Fixation marker sizing", 

1405 category=CATEGORY_DISPLAY, 

1406 summary="Marker size encodes fixation duration on one fixed scale.", 

1407 formula=( 

1408 "Fixed scales (`marker_size_scale` = `sqrt`, the default; `linear`; " 

1409 "`log`): `size = s_min + (s_max − s_min) · (f(d) − f(lo)) / " 

1410 "(f(hi) − f(lo))`, with `d` clamped to the duration bounds " 

1411 "`[lo, hi]` (`marker_duration_range`, default 50–600 ms) and `f` " 

1412 "= √, identity or ln. `relative`: linear between the drawn set's " 

1413 "own shortest and longest duration (the scale before the fixed one; " 

1414 "older saved configs and Share links keep it). **Display only** — " 

1415 "never a recorded value." 

1416 ), 

1417 code="scanpath_studio/plots.py:_compute_marker_sizes", 

1418 unit="px (marker diameter)", 

1419 missing="A missing duration is treated as 0 ms: the smallest marker.", 

1420 precedence=( 

1421 "One scale for single-trial figures, both comparison sides, replays " 

1422 "and bulk exports, so a duration draws at one size in all of them; " 

1423 "only the px range is per scanpath in Compare." 

1424 ), 

1425 tiers="C, D", 

1426 status=STATUS_CONVENTION, 

1427 consumers=(_UI, _API, _CLI, _EXPORT), 

1428 tests=( 

1429 "tests/test_duration_scale.py", 

1430 "tests/test_plots.py", 

1431 "tests/test_builder_parity.py", 

1432 ), 

1433 ), 

1434 Computation( 

1435 id="disp.axis_ranges", 

1436 name="Axis ranges and inversion", 

1437 category=CATEGORY_DISPLAY, 

1438 summary="Screen coordinates, drawn the way the screen is.", 

1439 formula=( 

1440 "The y axis is inverted (`y_range = [max, min]`) so the figure " 

1441 "matches the display; ranges come from the canvas, not the data, " 

1442 "when a canvas size is known." 

1443 ), 

1444 code="scanpath_studio/plots.py:_compute_axis_ranges", 

1445 unit="px", 

1446 tiers="C, D", 

1447 status=STATUS_CONVENTION, 

1448 consumers=(_UI, _API, _CLI, _EXPORT), 

1449 tests=("tests/test_plots.py",), 

1450 ), 

1451 Computation( 

1452 id="disp.true_scale", 

1453 name="True-scale text rendering", 

1454 category=CATEGORY_DISPLAY, 

1455 summary="One line of text fills its share of the recorded line pitch.", 

1456 formula=( 

1457 "A word label's font is `1/line_spacing` of the line pitch (the median " 

1458 "line-to-line distance of the word boxes), capped so the words fit " 

1459 "their box widths (`plots._width_fit_font`; the smaller wins), in " 

1460 "data pixels converted at the figure's display scale. When the " 

1461 "boxes are monospace words padded alike — half the gap to each " 

1462 "neighbour — the font is read off them instead: one character cell " 

1463 "is the slope of box width over word length " 

1464 "(`plots._padded_monospace_font`), over the font's advance. The " 

1465 "figure is drawn at its exact pixel size and scaled as one block." 

1466 ), 

1467 code="scanpath_studio/tabs.py:_render_true_scale_chart", 

1468 # The spatial plot must stay on this path: `st.plotly_chart` loses the 

1469 # scale guarantee (a developer rule, so not published). 

1470 tiers="D", 

1471 status=STATUS_CONVENTION, 

1472 consumers=(_UI,), 

1473 tests=("tests/test_plots.py",), 

1474 ), 

1475 Computation( 

1476 id="disp.animation_timing", 

1477 name="Animation timing", 

1478 category=CATEGORY_DISPLAY, 

1479 summary="How recorded time maps to playback time.", 

1480 formula=( 

1481 "Frames sit on a uniform reading-time grid over `fix.rebased_onsets`; " 

1482 "the frame at reading time t is on screen once t / playback speed of " 

1483 "wall time has passed, so a replay lasts reading span / speed. The " 

1484 "player keeps that clock itself, skipping frames a display is too " 

1485 "slow to show, and a GIF/MP4 lasts the same. A multipart replay " 

1486 "changes screen at the boundary and draws no connector across " 

1487 "canvases." 

1488 ), 

1489 code="scanpath_studio/plots.py:make_scanpath_animation", 

1490 unit="ms (recorded) → ms (playback)", 

1491 precedence=( 

1492 "Plotly's own frame queue is never the clock: it rounds every frame " 

1493 "up to whole display ticks and the error accumulates (BUG-93). " 

1494 "Without the player (`fig.show()`) the figure falls back to it." 

1495 ), 

1496 tiers="C, D", 

1497 status=STATUS_CONVENTION, 

1498 consumers=(_UI, _API, _CLI, _EXPORT), 

1499 tests=("tests/test_replay_player.py", "tests/test_animation_export.py"), 

1500 ), 

1501 Computation( 

1502 id="disp.illustration", 

1503 name="Illustration disclosure", 

1504 category=CATEGORY_DISPLAY, 

1505 summary="When a figure stops being a faithful record.", 

1506 formula=( 

1507 "Views that no longer show the data as recorded — snapped " 

1508 "fixations, arced saccades, hidden or windowed fixations, a replay " 

1509 "not at real time, an authored scanpath — are labeled *Illustration*." 

1510 ), 

1511 code="scanpath_studio/illustration.py:illustration_reasons", 

1512 tiers="C, D", 

1513 status=STATUS_VERIFIED, 

1514 consumers=(_UI, _API, _CLI, _EXPORT), 

1515 tests=("tests/test_illustration.py", "tests/test_disclosure.py"), 

1516 ), 

1517) 

1518 

1519BY_ID = {entry.id: entry for entry in REGISTER} 

1520 

1521#: Entries the default build does not compute: they need 

1522#: ``SCANPATH_EXPERIMENTAL=1`` (`constants.computed_measures_enabled`, 

1523#: `preprocessing_enabled`, `drift_correction_enabled`, `similarity_enabled`). 

1524#: The page marks them, so it does not present held-back work as shipped. 

1525EXPERIMENTAL_IDS = frozenset( 

1526 { 

1527 "measure.ffd", 

1528 "measure.fprt", 

1529 "measure.rpd", 

1530 "measure.tfd", 

1531 "measure.nfix", 

1532 "measure.skip", 

1533 "measure.regressions", 

1534 "measure.landing_position", 

1535 "measure.landing_distance", 

1536 "measure.second_pass", 

1537 "measure.single_fix", 

1538 "measure.reg_in_count", 

1539 "pre.merge_short", 

1540 "pre.exclude_short", 

1541 "pre.blink_adjacent", 

1542 "pre.cleaning_report", 

1543 "pre.sentence_measures", 

1544 "pre.saccade_table", 

1545 "pre.character_grid", 

1546 "pre.sensitivity", 

1547 "align.algorithms", 

1548 "agg.reader_summary", 

1549 "agg.trial_summary", 

1550 "agg.landing_curve", 

1551 "sim.nld", 

1552 "sim.aoi_sequence", 

1553 "sim.windowed", 

1554 } 

1555) 

1556 

1557_TRACKER_ID = r"#?[A-Z]{2,5}-\d+" 

1558_TRACKER_PATTERNS = ( 

1559 # "(BUG-25)", "(#PRE-21)", "(PRE-11/12)", "(BUG-61..66)", "(see VIZ-8)" 

1560 re.compile( 

1561 rf"\s*\((?:see |cf\. )?{_TRACKER_ID}" 

1562 rf"(?:\s*(?:[,/]|\.\.|and|&)\s*(?:{_TRACKER_ID}|\d+))*\)" 

1563 ), 

1564 # "only with SCANPATH_EXPERIMENTAL=1 — PRE-22)", "blanked, BUG-63)" 

1565 re.compile(rf"\s*[—–,-]\s*{_TRACKER_ID}(?=\))"), 

1566 # a sentence opening "#BUG-27: " or "#BUG-27 — " 

1567 re.compile(rf"(?:^|(?<=\. )){_TRACKER_ID}(?::| —)\s+"), 

1568) 

1569 

1570 

1571def _public(text: str) -> str: 

1572 """The register's text as the published page shows it: tracker ids are 

1573 for the code and its history, not for a reader of the docs.""" 

1574 *inline, opening = _TRACKER_PATTERNS 

1575 for pattern in inline: 

1576 text = pattern.sub("", text) 

1577 # Only a sentence whose opening id was dropped gets a capital again — 

1578 # column names (`trial_id`) and formulas (`x .. x + width`) keep their case. 

1579 text = opening.sub("\0", text) 

1580 return re.sub("\0([a-z]?)", lambda m: m.group(1).upper(), text) 

1581 

1582 

1583def _experimental_note(entry: Computation) -> list[str]: 

1584 if entry.id not in EXPERIMENTAL_IDS: 

1585 return [] 

1586 if entry.category == CATEGORY_MEASURE: 

1587 body = ( 

1588 "Scanpath Studio does not compute this in this release. A value " 

1589 "your dataset brings is shown as given, defined by the software " 

1590 "that exported it." 

1591 ) 

1592 else: 

1593 body = "Not in this release." 

1594 return ['!!! warning "Experimental"', "", f" {body}", ""] 

1595 

1596 

1597def entries_in(category: str) -> tuple[Computation, ...]: 

1598 """Every register entry in one category, in declaration order.""" 

1599 return tuple(entry for entry in REGISTER if entry.category == category) 

1600 

1601 

1602def measure_entry(column: str) -> Computation | None: 

1603 """The reading-measure or fixation entry that defines ``column`` — the one 

1604 whose ``output`` names it — or ``None`` for a value the register does not 

1605 derive (a fixation's recorded duration). Lets the app quote the register's 

1606 own summary and unit beside a measure instead of keeping a second copy.""" 

1607 for entry in REGISTER: 

1608 if not entry.id.startswith(("measure.", "fix.")): 

1609 continue 

1610 if column in (part.strip() for part in entry.output.split(",")): 

1611 return entry 

1612 return None 

1613 

1614 

1615def anchor(entry_id: str) -> str: 

1616 """The page anchor of one entry — its id, so a link to ``measure.ffd`` 

1617 survives any rewording of the entry's name (``#measure-ffd``).""" 

1618 return entry_id.replace(".", "-").replace("_", "-") 

1619 

1620 

1621def to_markdown() -> str: 

1622 """Render the register as the ``docs/computations.md`` page. 

1623 

1624 Generated rather than hand-maintained so the documentation cannot drift 

1625 from the catalogue the integrity tests check. 

1626 """ 

1627 lines: list[str] = [ 

1628 "<!-- Generated by `python -m scanpath_studio.computations`. Do not edit. -->", 

1629 "", 

1630 "# Computations & methodology", 

1631 "", 

1632 f"Register version **{REGISTER_VERSION}** · " 

1633 f"{len(REGISTER)} entries across {len(CATEGORIES)} categories.", 

1634 "", 

1635 "Every operation that derives or semantically changes a value you can " 

1636 "see, export, or fetch through the API is listed here with its formula, " 

1637 "its units, and how far it has actually been verified. Pure layout and " 

1638 "byte-preserving file I/O are out of scope; filtering, precedence and " 

1639 "assignment are in, because they change *which observations* a result " 

1640 "stands for.", 

1641 "", 

1642 "## How to read the status column", 

1643 "", 

1644 "| Status | Means |", 

1645 "| --- | --- |", 

1646 "| **Verified** | A hand-calculated oracle or exact invariant exists and passes. |", 

1647 "| **Partially verified** | Tested, but without an independent reference implementation. |", 

1648 "| **Unverified** | Exercised by tests only for execution, not for meaning. |", 

1649 "| **Intentional convention** | A choice that can only be documented, not proved. |", 

1650 "", 

1651 "Verification tiers: **A** hand-calculated synthetic oracle · **B** " 

1652 "independent reference implementation · **C** property/invariant tests · " 

1653 "**D** cross-surface parity (UI, API, CLI, export agree).", 

1654 "", 

1655 '!!! note "Tier B is largely absent, on purpose"', 

1656 "", 

1657 " Comparing against an independent implementation " 

1658 "[is planned](https://github.com/lacclab/scanpath-studio/issues/130). " 

1659 "Scientific measures therefore read *Partially verified* even where " 

1660 "their hand oracle is exact.", 

1661 "", 

1662 "Entries marked *experimental* are not in this release. They are " 

1663 "listed so that their definitions are on record.", 

1664 "", 

1665 "## Summary", 

1666 "", 

1667 "| ID | Name | Category | Unit | Status |", 

1668 "| --- | --- | --- | --- | --- |", 

1669 ] 

1670 for entry in REGISTER: 

1671 status = entry.status + ( 

1672 " · experimental" if entry.id in EXPERIMENTAL_IDS else "" 

1673 ) 

1674 lines.append( 

1675 f"| [`{entry.id}`](#{anchor(entry.id)}) | {entry.name} | " 

1676 f"{entry.category} | {entry.unit or '—'} | {status} |" 

1677 ) 

1678 lines.append("") 

1679 for category in CATEGORIES: 

1680 entries = entries_in(category) 

1681 if not entries: 

1682 continue 

1683 lines += [f"## {category}", ""] 

1684 for entry in entries: 

1685 lines += [ 

1686 f"### `{entry.id}` — {entry.name} {{ #{anchor(entry.id)} }}", 

1687 "", 

1688 *_experimental_note(entry), 

1689 _public(entry.summary), 

1690 "", 

1691 f"**Formula.** {_public(entry.formula)}", 

1692 "", 

1693 ] 

1694 rows = [ 

1695 ("Output", _public(entry.output)), 

1696 ("Unit", entry.unit), 

1697 ("Grouping / ordering", _public(entry.grouping)), 

1698 ("Missing & edge cases", _public(entry.missing)), 

1699 ("Precedence & caveats", _public(entry.precedence)), 

1700 ("Reference", _public(entry.reference)), 

1701 ("Code", f"`{entry.code}`"), 

1702 ("Consumers", ", ".join(_public(c) for c in entry.consumers)), 

1703 ("Tests", ", ".join(f"`{t}`" for t in entry.tests)), 

1704 ("Verification", f"tier {entry.tiers} — **{entry.status}**"), 

1705 ] 

1706 lines += ["| | |", "| --- | --- |"] 

1707 for label, value in rows: 

1708 if value: 

1709 lines.append(f"| **{label}** | {value} |") 

1710 lines.append("") 

1711 return "\n".join(lines).rstrip() + "\n" 

1712 

1713 

1714def main() -> None: # pragma: no cover - thin CLI wrapper 

1715 from pathlib import Path 

1716 

1717 target = Path(__file__).resolve().parents[1] / "docs" / "computations.md" 

1718 target.write_text(to_markdown(), encoding="utf-8") 

1719 print(f"Wrote {target} ({len(REGISTER)} entries)") 

1720 

1721 

1722if __name__ == "__main__": # pragma: no cover 

1723 main()