Coverage for scanpath_studio/computations.py: 100%
102 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""The computation register (VAL-5).
3Every operation that **derives or semantically changes a value** a user, an
4export or an API consumer can see, recorded once with its formula, its units,
5its missing-data behaviour, and how far it has actually been verified. Pure UI
6layout and byte-preserving file I/O are out of scope; filtering, precedence and
7assignment are in, because they change *which observations* a result stands for
8even when no arithmetic happens.
10This module is data, not behaviour. It exists so that
12* ``tests/test_computations.py`` can assert the catalogue has not drifted from
13 the code — every ``aggregation.MEASURES`` entry, every drift-correction
14 algorithm and every similarity metric must map to an entry here, and every
15 entry must point at a function that exists;
16* ``docs/computations.md`` can be generated from one source rather than
17 maintained in parallel with it (``python -m scanpath_studio.computations``).
19**Verification tiers.** ``A`` a hand-calculated synthetic oracle · ``B`` an
20independent reference implementation or corpus comparison · ``C`` property /
21invariant tests · ``D`` cross-surface parity (UI, API, CLI, export agree).
23**Status is deliberately conservative.** *Verified* means a semantic oracle
24exists and passes — not that a line of code was executed. Tier B is
25systematically absent: it needs an independent implementation to compare
26against, which is #VAL-4, on hold at the user's request until the register
27itself has been read. Every scientific measure therefore reads *Partially
28verified* even where its hand oracle is exact, and that is the honest state of
29the world rather than a gap to paper over. *Convention* marks a choice that
30cannot be right or wrong, only documented — a display transform, a tie-break, a
31default threshold.
32"""
34from __future__ import annotations
36import re
37from dataclasses import dataclass, field
39#: Bumped when an entry's *meaning* changes (a formula, a unit, a default), not
40#: when prose is edited. Exported alongside results so a bundle can name the
41#: methodology it was produced under.
42REGISTER_VERSION = "2"
44CATEGORY_IMPORTED = "Imported / precomputed"
45CATEGORY_NORMALIZATION = "Normalization / inference"
46CATEGORY_ASSIGNMENT = "Assignment / classification"
47CATEGORY_PREPROCESSING = "Preprocessing"
48CATEGORY_MEASURE = "Scientific measure"
49CATEGORY_AGGREGATION = "Statistical aggregation"
50CATEGORY_SIMILARITY = "Similarity"
51CATEGORY_GEOMETRY = "Unit / coordinate conversion"
52CATEGORY_DISPLAY = "Display / export transformation"
54CATEGORIES: tuple[str, ...] = (
55 CATEGORY_IMPORTED,
56 CATEGORY_NORMALIZATION,
57 CATEGORY_ASSIGNMENT,
58 CATEGORY_PREPROCESSING,
59 CATEGORY_MEASURE,
60 CATEGORY_AGGREGATION,
61 CATEGORY_SIMILARITY,
62 CATEGORY_GEOMETRY,
63 CATEGORY_DISPLAY,
64)
66STATUS_VERIFIED = "Verified"
67STATUS_PARTIAL = "Partially verified"
68STATUS_UNVERIFIED = "Unverified"
69STATUS_CONVENTION = "Intentional convention"
71STATUSES: tuple[str, ...] = (
72 STATUS_VERIFIED,
73 STATUS_PARTIAL,
74 STATUS_UNVERIFIED,
75 STATUS_CONVENTION,
76)
79@dataclass(frozen=True)
80class Computation:
81 """One derived value, with everything needed to reproduce and judge it."""
83 id: str
84 name: str
85 category: str
86 summary: str
87 formula: str
88 code: str
89 output: str = ""
90 unit: str = ""
91 grouping: str = ""
92 missing: str = ""
93 precedence: str = ""
94 tiers: str = ""
95 status: str = STATUS_UNVERIFIED
96 reference: str = ""
97 consumers: tuple[str, ...] = field(default_factory=tuple)
98 tests: tuple[str, ...] = field(default_factory=tuple)
100 @property
101 def module(self) -> str:
102 return self.code.split(":", 1)[0]
104 @property
105 def symbol(self) -> str:
106 return self.code.split(":", 1)[1] if ":" in self.code else ""
109_UI = "UI"
110_API = "API"
111_CLI = "CLI"
112_EXPORT = "Export"
113_CORPUS = "Corpus Analysis"
114_INSPECT = "Data Management"
115# Surfaces held back from the app this release behind `SCANPATH_EXPERIMENTAL=1`,
116# named as such so the register does not advertise a panel a user cannot open.
117_UI_PREPROCESSING = "UI (Preprocessing panel — not in this release, PRE-22)"
118_INSPECT_DERIVED = "Data Management (derived tables — not in this release, UX-126)"
119_API_EXPERIMENTAL = "API (not in this release, PRE-21)"
120# FFD / FPRT / RPD / single-fixation duration, on both paths: computed
121# (`measures.compute_per_word_measures`) and imported
122# (`data._blank_unfixated_measures`, #374).
123_UNFIXATED_MISSING = (
124 "Never fixated ⇒ NaN, not 0, so a skipped word is left out of every mean. "
125 "An imported 0 is blanked too, wherever the word's fixation count is 0 — "
126 "or, with no count mapped, its total fixation duration is 0 (BUG-63)."
127)
130REGISTER: tuple[Computation, ...] = (
131 # ------------------------------------------------------------------
132 # Normalization / inference
133 # ------------------------------------------------------------------
134 Computation(
135 id="norm.words",
136 name="Word table normalization",
137 category=CATEGORY_NORMALIZATION,
138 summary="Map an arbitrary word/IA export onto the canonical word columns.",
139 formula=(
140 "For each canonical field, `pick_column` walks a candidate list and "
141 "takes the first column that exists; the user's mapping overrides it. "
142 "Unmapped optional fields are dropped unless listed in "
143 "`WORD_OPTIONAL_FIELDS`."
144 ),
145 code="scanpath_studio/data.py:normalize_words",
146 output="participant_id, trial_id, text_id, word_id, text, x, y, width, height",
147 precedence="An explicit user mapping always beats auto-detection.",
148 missing="A missing *required* field raises with the columns it looked for.",
149 tiers="C, D",
150 status=STATUS_PARTIAL,
151 consumers=(_UI, _API, _CLI, _EXPORT),
152 tests=("tests/test_data.py", "tests/test_column_mapping.py"),
153 ),
154 Computation(
155 id="norm.fixations",
156 name="Fixation table normalization",
157 category=CATEGORY_NORMALIZATION,
158 summary="Map an arbitrary fixation report onto the canonical columns.",
159 formula=(
160 "As `norm.words`, over the fixation candidate lists. `order_in_trial` "
161 "is assigned by sorting each trial on `timestamp_ms`; `fixation_id` is "
162 "synthesized per trial when the export carries none."
163 ),
164 code="scanpath_studio/data.py:normalize_fixations",
165 output="participant_id, trial_id, x, y, duration_ms, timestamp_ms, …",
166 grouping="(participant_id, trial_id[, screen_id])",
167 missing="Rows with no coordinates survive when a word/AoI id is mapped.",
168 tiers="C, D",
169 status=STATUS_PARTIAL,
170 consumers=(_UI, _API, _CLI, _EXPORT),
171 tests=("tests/test_data.py",),
172 ),
173 Computation(
174 id="norm.box_edges",
175 name="Word box from edges",
176 category=CATEGORY_NORMALIZATION,
177 summary="Convert EyeLink IA edges to origin+size.",
178 formula=(
179 "x = IA_LEFT · y = IA_TOP · width = IA_RIGHT − IA_LEFT · "
180 "height = IA_BOTTOM − IA_TOP."
181 ),
182 code="scanpath_studio/data.py:normalize_words",
183 output="x, y, width, height",
184 unit="px (screen coordinates, y increasing downwards)",
185 tiers="A, C",
186 status=STATUS_VERIFIED,
187 consumers=(_UI, _API, _CLI, _EXPORT),
188 tests=("tests/test_word_box_geometry.py",),
189 ),
190 Computation(
191 id="norm.trial_id_composite",
192 name="Composite trial identity",
193 category=CATEGORY_NORMALIZATION,
194 summary="Build one unique trial id from several columns.",
195 formula=(
196 "The mapped Trial ID columns are joined in the order given, "
197 "separated by `_`, after casting each to string. A `_` or `\\` "
198 "inside a part is escaped with a `\\` first, so two different "
199 "tuples never give the same id; parts with neither compose as a "
200 "plain join."
201 ),
202 code="scanpath_studio/data.py:trial_id_series",
203 output="trial_id",
204 missing="A row missing any component keeps the literal string of that part.",
205 tiers="C, D",
206 status=STATUS_PARTIAL,
207 consumers=(_UI, _API, _CLI),
208 tests=("tests/test_trial_identity.py", "tests/test_composite_ids.py"),
209 ),
210 Computation(
211 id="norm.flags",
212 name="Flag coercion",
213 category=CATEGORY_NORMALIZATION,
214 summary="Read EyeLink's string booleans as booleans (BUG-7).",
215 formula=(
216 "Numbers go by `!= 0`. Strings are matched case-insensitively "
217 "against `{'', '.', '0', '0.0', 'false', 'f', 'no', 'n', 'na', "
218 "'nan', '-'}` → False; anything else → True."
219 ),
220 code="scanpath_studio/data.py:coerce_flag",
221 output="bool",
222 missing=(
223 "NaN → False for an operational flag (blink, excluded). A supplied "
224 "reading-measure flag (skip, regression in / out) keeps it missing "
225 "instead — `''`, `'.'`, `'na'`, `'nan'`, `'-'` and NaN read as NA "
226 "(`coerce_measure_flag`, a nullable boolean)."
227 ),
228 tiers="A, C",
229 status=STATUS_VERIFIED,
230 reference="Guards the `'.'`-as-missing convention in EyeLink IA reports.",
231 consumers=(_UI, _API, _CLI, _EXPORT),
232 tests=("tests/test_data.py",),
233 ),
234 Computation(
235 id="norm.stimulus_broadcast",
236 name="Stimulus-level word broadcast",
237 category=CATEGORY_NORMALIZATION,
238 summary="Share one stimulus' word boxes across every participant who read it.",
239 formula=(
240 "Words with no participant column are copied once per reading "
241 "(participant × trial [× screen]) in the fixations, stamped with "
242 "that reading's ids. Per reading, the boxes are those of the first "
243 "words trial found by its trial ID, then its trial ID before a "
244 "repeat's _r2 suffix, then its Text ID (only a Text ID the fixations "
245 "map, and never one the words give to more than one trial). A "
246 "trial-ID match always stands; mapped Text IDs that disagree with "
247 "it are warned about (DATA-49)."
248 ),
249 code="scanpath_studio/data.py:broadcast_stimulus_words",
250 missing=(
251 "No fixations for a text ⇒ its words are not broadcast. Some "
252 "readings unmatched ⇒ StimulusJoinWarning with the counts; none "
253 "matched, or any multipart screen unmatched ⇒ StimulusJoinError, "
254 "never an empty table."
255 ),
256 tiers="C",
257 status=STATUS_PARTIAL,
258 consumers=(_UI, _API, _CLI),
259 tests=("tests/test_stimulus_join.py", "tests/test_dataset_support.py"),
260 ),
261 Computation(
262 id="norm.aoi_center_placement",
263 name="AoI-only fixation placement",
264 category=CATEGORY_NORMALIZATION,
265 summary="Place a fixation with no x/y at its word box's center.",
266 formula="x = word.x + width/2 · y = word.y + height/2.",
267 code="scanpath_studio/data.py:harmonize_frames",
268 unit="px",
269 precedence="Only when x/y are absent; recorded coordinates always win.",
270 missing="No matching word box ⇒ the fixation keeps no coordinates.",
271 tiers="A, C",
272 status=STATUS_VERIFIED,
273 consumers=(_UI, _API, _CLI),
274 tests=("tests/test_data.py",),
275 ),
276 Computation(
277 id="norm.participant_metadata",
278 name="Participant metadata join",
279 category=CATEGORY_NORMALIZATION,
280 summary="Attach a participant-level table without broadcasting it (DATA-20).",
281 formula=(
282 "Left join on string `participant_id`. Duplicate ids that agree "
283 "are combined field by field (each field keeps the one non-missing "
284 "value the rows hold); duplicate ids that **disagree** are dropped "
285 "and reported, "
286 "so no `groupby.first()` winner is ever invented. A field is "
287 "projected onto the per-trial frame, never onto word/fixation rows."
288 ),
289 code="scanpath_studio/metadata.py:build_participant_metadata",
290 output="One column per registered field, at participant grain",
291 missing="A participant with no row reads as missing everywhere, never as a default.",
292 precedence="A real recorded column of the same name always wins.",
293 tiers="A, C, D",
294 status=STATUS_VERIFIED,
295 consumers=(_UI, _API, _CLI, _EXPORT, _INSPECT),
296 tests=("tests/test_metadata.py", "tests/test_metadata_duplicates.py"),
297 ),
298 # ------------------------------------------------------------------
299 # Assignment / classification
300 # ------------------------------------------------------------------
301 Computation(
302 id="assign.fixation_to_word",
303 name="Fixation → word assignment",
304 category=CATEGORY_ASSIGNMENT,
305 summary="The single highest-risk step: which word a fixation counts for.",
306 formula=(
307 "1. Bounding-box containment against the trial's word boxes — the "
308 "experiment's own rectangles (`geom.word_box_bounds`), so on a "
309 "tiling corpus a fixation on the space *after* a word is credited to "
310 "that word, as EyeLink's interest-area report credits it. Boxes are "
311 "half-open, `x0 ≤ x < x1` and `y0 ≤ y < y1` "
312 "(`measures.word_box_contains`), so a point on an edge two boxes "
313 "share goes to the one that starts there — the next word, the line "
314 "below — as EyeLink assigns it. "
315 "2. Otherwise `word_id = NaN` (out of text). There is no snapping "
316 "to a nearby word."
317 ),
318 code="scanpath_studio/measures.py:assign_fixations_to_words",
319 output="word_id",
320 grouping="(participant_id, trial_id[, screen_id]) — never across screens",
321 missing="Unassignable fixations keep NaN and are excluded from word measures.",
322 precedence=(
323 "Runs only when the fixations carry no word id. A mapped `word_id` "
324 "(on the bundled demo, EyeLink's `CURRENT_FIX_INTEREST_AREA_ID`) is "
325 "used exactly as given, blanks included — nothing is computed and "
326 "no blank is filled — unless `overwrite=True`. #BUG-83: geometry "
327 "agrees with that column on all 3,208 of the demo's EyeLink-assigned "
328 "fixations."
329 ),
330 tiers="A, C",
331 status=STATUS_PARTIAL,
332 consumers=(_UI, _API, _CLI, _EXPORT, _CORPUS),
333 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
334 ),
335 Computation(
336 id="assign.in_text",
337 name="Out-of-text flag",
338 category=CATEGORY_ASSIGNMENT,
339 summary="Whether a fixation landed on any word of the stimulus.",
340 formula=(
341 "The fixation falls inside some word box (`word_box_bounds`, tested "
342 "half-open by `word_box_contains`, as `assign.fixation_to_word` "
343 "tests it). Box containment only, so a fixation the data's own "
344 "`word_id` puts on a word but that lies outside every box still "
345 "counts as out-of-text."
346 ),
347 code="scanpath_studio/measures.py:fixation_in_text_mask",
348 output="bool mask",
349 tiers="A, C",
350 status=STATUS_VERIFIED,
351 consumers=(_UI, _API, _CORPUS),
352 tests=("tests/test_synthetic.py",),
353 ),
354 Computation(
355 id="assign.line_cluster",
356 name="Visual line clustering",
357 category=CATEGORY_ASSIGNMENT,
358 summary="Derive text lines from word-box geometry, not from `line_idx`.",
359 formula=(
360 "Word boxes are sorted by `y` and split wherever the gap between "
361 "consecutive centers exceeds `tol_frac` (0.5) of the median box "
362 "height. Exists because `line_idx` is a constant in many IA exports."
363 ),
364 code="scanpath_studio/measures.py:cluster_word_lines",
365 output="Line index per word",
366 tiers="A, C",
367 status=STATUS_PARTIAL,
368 consumers=(_UI, _API, _CORPUS),
369 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
370 ),
371 Computation(
372 id="assign.runs",
373 name="Runs and passes",
374 category=CATEGORY_ASSIGNMENT,
375 summary="Trial run, line run, and per-word visit/pass indices (PRE-16).",
376 formula=(
377 "Consecutive fixations on the same word form one *visit*; the n-th "
378 "visit to a word is its n-th pass. Line runs break whenever the "
379 "assigned line changes."
380 ),
381 code="scanpath_studio/measures.py:materialize_runs",
382 output=(
383 "run, linerun, word_runid, word_run (the visit's pass number), "
384 "word_run_fix, nrun, reread (word_run > 1)"
385 ),
386 grouping="Ordered by `timestamp_ms` within a trial",
387 precedence=(
388 "Always recomputed: an imported column under any of these names is "
389 "replaced. An imported `pass_index` (EyeLink's `reread` is renamed "
390 "to it on load) is a separate column and is kept as given — "
391 "nothing computes `pass_index`."
392 ),
393 tiers="A, C",
394 status=STATUS_PARTIAL,
395 consumers=(_UI, _API, _EXPORT, _CORPUS),
396 tests=("tests/test_measures.py",),
397 ),
398 Computation(
399 id="assign.progression",
400 name="Progression and regression flags",
401 category=CATEGORY_ASSIGNMENT,
402 summary="Whether the *outgoing* saccade moves forward in the text.",
403 formula=(
404 "`progression = sign(next word_id − word_id)`. "
405 "`is_regression = word_id < running max word_id in the trial` — i.e. "
406 "relative to the furthest word reached, not to the previous fixation."
407 ),
408 code="scanpath_studio/measures.py:enrich_fixations",
409 output="progression ∈ {−1, 0, 1}, is_regression",
410 grouping="Per trial, in timestamp order",
411 missing="Unassigned fixations give progression 0.",
412 tiers="A, C",
413 status=STATUS_VERIFIED,
414 consumers=(_UI, _API, _EXPORT, _CORPUS),
415 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
416 ),
417 Computation(
418 id="assign.saccade_class",
419 name="Saccade reading class",
420 category=CATEGORY_ASSIGNMENT,
421 summary="Label each outgoing saccade by its reading role (VIZ-8).",
422 formula=(
423 "From the word and text line of the two fixations, in this order: "
424 "refixation (same word), regression (up to an earlier line, or back "
425 "within a line), return sweep (down to a later line), forward (the "
426 "next word on the line), skip (two or more words ahead on the line); "
427 "`other` when either fixation has no assigned word."
428 ),
429 code="scanpath_studio/measures.py:classify_saccades",
430 output="(not stored — computed for each figure)",
431 precedence=(
432 "Always computed; an imported `saccade_type` / `NEXT_SAC_DIRECTION` "
433 "(a direction) is not used."
434 ),
435 tiers="A, C",
436 status=STATUS_PARTIAL,
437 consumers=(_UI, _API, _EXPORT),
438 tests=("tests/test_saccade_class_filter.py",),
439 ),
440 # ------------------------------------------------------------------
441 # Scientific measures
442 # ------------------------------------------------------------------
443 Computation(
444 id="measure.ffd",
445 name="First fixation duration (FFD)",
446 category=CATEGORY_MEASURE,
447 summary="Duration of the first fixation on a word.",
448 formula=(
449 "Duration of the word's first fixation, whenever it comes — as "
450 "EyeLink's `IA_FIRST_FIXATION_DURATION`, so a computed and an "
451 "imported value mean the same. Not conditioned on first pass: a word "
452 "first reached by a regression has an FFD and `skip_flag = True`; "
453 "filter on `skip_flag` for first-pass-only analyses."
454 ),
455 code="scanpath_studio/measures.py:compute_per_word_measures",
456 output="first_fixation_ms",
457 unit="ms",
458 grouping="(participant, trial, word)",
459 missing=_UNFIXATED_MISSING,
460 precedence="A precomputed `IA_FIRST_FIXATION_DURATION` wins.",
461 tiers="A, D",
462 status=STATUS_PARTIAL,
463 reference="Rayner (1998), standard reading-measure definitions.",
464 consumers=(_UI, _API),
465 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
466 ),
467 Computation(
468 id="measure.fprt",
469 name="First-pass gaze duration (FPRT)",
470 category=CATEGORY_MEASURE,
471 summary="Sum of the fixations in the word's first visit.",
472 formula=(
473 "Sum of every fixation in the word's **first** run, i.e. before the "
474 "gaze leaves the word for the first time — whenever that run starts "
475 "(EyeLink's `IA_FIRST_RUN_DWELL_TIME`; not conditioned on first pass, "
476 "as `measure.ffd`). A fixation outside every word ends the run "
477 "(BUG-66), as it does for `measure.second_pass`."
478 ),
479 code="scanpath_studio/measures.py:compute_per_word_measures",
480 output="first_pass_gaze_duration_ms",
481 unit="ms",
482 grouping="(participant, trial, word)",
483 missing=_UNFIXATED_MISSING,
484 precedence="A precomputed IA gaze duration wins.",
485 tiers="A, D",
486 status=STATUS_PARTIAL,
487 reference="Rayner (1998).",
488 consumers=(_UI, _API),
489 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
490 ),
491 Computation(
492 id="measure.rpd",
493 name="Regression-path duration (RPD / go-past)",
494 category=CATEGORY_MEASURE,
495 summary="First entry to the word until the gaze passes it to the right.",
496 formula=(
497 "Total time from the word's first fixation until the first fixation "
498 "on a **later** word — every fixation in between, including a first "
499 "visit to an earlier, skipped word during the regression (BUG-61). "
500 "Matches EyeLink's `IA_REGRESSION_PATH_DURATION` on 1779 of the "
501 "bundled demo's 1780 fixated words. Fixations outside every word "
502 "neither extend nor close the window."
503 ),
504 code="scanpath_studio/measures.py:compute_per_word_measures",
505 output="regression_path_duration_ms",
506 unit="ms",
507 grouping="(participant, trial, word)",
508 missing=_UNFIXATED_MISSING,
509 tiers="A",
510 status=STATUS_PARTIAL,
511 reference=(
512 "Definitions differ across toolkits (go-past vs regression path); "
513 "`eyekit` is the intended comparison. Unresolved until that "
514 "cross-validation runs."
515 ),
516 consumers=(_UI, _API),
517 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
518 ),
519 Computation(
520 id="measure.tfd",
521 name="Total fixation duration (TFD)",
522 category=CATEGORY_MEASURE,
523 summary="All time spent on a word across the whole trial.",
524 formula="Sum of every fixation assigned to the word, any pass.",
525 code="scanpath_studio/measures.py:compute_per_word_measures",
526 output="total_fixation_duration_ms",
527 unit="ms",
528 missing="Never fixated ⇒ 0 (the word *was* read past; it got no time).",
529 precedence="A precomputed IA dwell time wins.",
530 tiers="A, D",
531 status=STATUS_PARTIAL,
532 consumers=(_UI, _API),
533 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
534 ),
535 Computation(
536 id="measure.nfix",
537 name="Fixations per word",
538 category=CATEGORY_MEASURE,
539 summary="Count of fixations assigned to a word.",
540 formula="Row count of the word's assigned fixations.",
541 code="scanpath_studio/measures.py:compute_per_word_measures",
542 output="n_fixations",
543 missing="Never fixated ⇒ 0.",
544 tiers="A",
545 status=STATUS_VERIFIED,
546 consumers=(_UI, _API),
547 tests=("tests/test_synthetic.py",),
548 ),
549 Computation(
550 id="measure.skip",
551 name="Skip flag / skip rate",
552 category=CATEGORY_MEASURE,
553 summary="Whether a word received no first-pass fixation.",
554 formula="`skip_flag = no fixation in the word's first pass`.",
555 code="scanpath_studio/measures.py:compute_per_word_measures",
556 output="skip_flag",
557 unit="rate when aggregated (0–1)",
558 missing="A word fixated only after a regression still counts as skipped.",
559 tiers="A",
560 status=STATUS_VERIFIED,
561 consumers=(_UI, _API),
562 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
563 ),
564 Computation(
565 id="measure.regressions",
566 name="Regression in/out flags",
567 category=CATEGORY_MEASURE,
568 summary="Whether a word was returned to, or left backwards.",
569 formula=(
570 "`regression_in_flag` — some later fixation lands on this word after "
571 "the gaze had moved past it. `regression_out_flag` — a regression "
572 "to an earlier word is made from this word during first pass, before "
573 "the eyes first leave it forwards (EyeLink's `IA_REGRESSION_OUT`, "
574 "BUG-64); a regression from it later in the trial does not count."
575 ),
576 code="scanpath_studio/measures.py:compute_per_word_measures",
577 output="regression_in_flag, regression_out_flag",
578 unit="rate when aggregated (0–1)",
579 precedence="Precomputed IA regression flags win (see `norm.flags`).",
580 tiers="A",
581 status=STATUS_PARTIAL,
582 consumers=(_UI, _API),
583 tests=("tests/test_measures.py", "tests/test_synthetic.py"),
584 ),
585 Computation(
586 id="measure.landing_position",
587 name="Initial landing position",
588 category=CATEGORY_MEASURE,
589 summary="Where in the word the first fixation landed, in letters.",
590 formula=(
591 "`char_width = geom.word_char_advance`; "
592 "`offset = first_fix_x − word.x` (LTR) or "
593 "`word.x + n·advance − first_fix_x` (RTL, BUG-27); "
594 "`landing_position = offset / char_width + 1` — so the first letter "
595 "starts at 1 and its center is 1.5. Unclipped: on a tiling corpus "
596 "the box's last cell is the space after the word, which belongs to "
597 "it (#BUG-83), so a first fixation there reads `n + 1` to `n + 2`."
598 ),
599 code="scanpath_studio/measures.py:compute_per_word_measures",
600 output="initial_landing_position",
601 unit="letters",
602 missing=(
603 "Never fixated, zero width, or no text ⇒ NaN. Measured from the "
604 "word's first fixation, first pass or not (as `measure.ffd`)."
605 ),
606 precedence=(
607 "VAL-5: the scale is `geom.word_char_advance`, not "
608 "`width / len(text)`, which on a tiling corpus puts every landing "
609 "~`(n+1)/n` too far into the word."
610 ),
611 tiers="A",
612 status=STATUS_PARTIAL,
613 reference=(
614 "Assumes a monospaced advance within the word box — exact for the "
615 "app's monospace default, approximate for proportional fonts."
616 ),
617 consumers=(_UI, _API),
618 tests=("tests/test_measures.py",),
619 ),
620 Computation(
621 id="measure.landing_distance",
622 name="Centred landing distance",
623 category=CATEGORY_MEASURE,
624 summary="Landing position relative to the word's center.",
625 formula=(
626 "`landing_position − (1 + len(text) / 2)` — the glyphs span "
627 "`[1, n + 1)`, so that is the word's center (BUG-65). The center of "
628 "the *letters*, not of the box: a tiling box's trailing space "
629 "(#BUG-83) would move it half a letter right."
630 ),
631 code="scanpath_studio/measures.py:compute_per_word_measures",
632 output="initial_landing_distance",
633 unit="letters (0 = word center, negative = left of center)",
634 missing="As `measure.landing_position`.",
635 tiers="A",
636 status=STATUS_PARTIAL,
637 consumers=(_UI, _API),
638 tests=("tests/test_measures.py",),
639 ),
640 Computation(
641 id="measure.second_pass",
642 name="Second-pass duration",
643 category=CATEGORY_MEASURE,
644 summary="Time spent on the word during its second visit.",
645 formula="Sum of the fixations in the word's second run.",
646 code="scanpath_studio/measures.py:compute_per_word_measures",
647 output="second_pass_duration_ms",
648 unit="ms",
649 missing=(
650 "Fewer than two runs ⇒ 0. An imported blank "
651 "`IA_SECOND_RUN_DWELL_TIME` becomes 0 too, where the fixation count "
652 "is known."
653 ),
654 tiers="A",
655 status=STATUS_PARTIAL,
656 consumers=(_UI, _API),
657 tests=("tests/test_measures.py",),
658 ),
659 Computation(
660 id="measure.single_fix",
661 name="Single-fixation duration",
662 category=CATEGORY_MEASURE,
663 summary="First-pass duration when the first pass was exactly one fixation.",
664 formula="FFD when the word's first run has length 1, else NaN.",
665 code="scanpath_studio/measures.py:compute_per_word_measures",
666 output="single_fixation_duration_ms",
667 unit="ms",
668 missing=("A first run of more than one fixation ⇒ NaN. " + _UNFIXATED_MISSING),
669 tiers="A",
670 status=STATUS_PARTIAL,
671 reference="Rayner (1998).",
672 consumers=(_UI, _API),
673 tests=("tests/test_measures.py",),
674 ),
675 Computation(
676 id="measure.reg_in_count",
677 name="Regressions into word",
678 category=CATEGORY_MEASURE,
679 summary="How many times the gaze came back to this word.",
680 formula=(
681 "Number of regressions into the word — entries from a later word "
682 "(EyeLink's `IA_REGRESSION_IN_COUNT`). A re-entry from an *earlier* "
683 "word is a new run but not a regression in."
684 ),
685 code="scanpath_studio/measures.py:compute_per_word_measures",
686 output="number_of_regressions_in",
687 missing="Never regressed into ⇒ 0.",
688 tiers="A",
689 status=STATUS_PARTIAL,
690 consumers=(_UI, _API),
691 tests=("tests/test_measures.py",),
692 ),
693 Computation(
694 id="fix.saccade_amplitude",
695 name="Saccade amplitude",
696 category=CATEGORY_MEASURE,
697 summary="Distance between consecutive fixations — always pixels (BUG-25).",
698 formula="`sqrt(dx² + dy²)` between consecutive fixations in the trial.",
699 code="scanpath_studio/measures.py:enrich_fixations",
700 output="saccade_amplitude",
701 unit="px",
702 grouping="Per trial, in timestamp order; the first fixation has none.",
703 missing="First fixation of a trial ⇒ NaN.",
704 precedence=(
705 "A source column literally named `saccade_amplitude` is assumed to "
706 "be pixels and kept. EyeLink's **degree**-valued "
707 "`NEXT_SAC_AMPLITUDE` / `PREVIOUS_SAC_AMPLITUDE` normalize to "
708 "`next_/prev_saccade_amplitude_deg` and never reach this column — "
709 "they are different quantities *and* different saccades."
710 ),
711 tiers="A, C",
712 status=STATUS_VERIFIED,
713 # The history (one column once meant px or deg) is in the changelog.
714 reference="(BUG-25)",
715 consumers=(_UI, _API, _EXPORT, _CORPUS),
716 tests=("tests/test_measures.py",),
717 ),
718 Computation(
719 id="fix.angles",
720 name="Saccade angles",
721 category=CATEGORY_MEASURE,
722 summary="Incoming and outgoing saccade direction.",
723 formula=(
724 "`angle_incoming = degrees(atan2(−dy, dx))` from the previous "
725 "fixation; `angle_outgoing` is the next fixation's incoming angle. "
726 "`−dy` because screen y grows downwards, so 0° is rightward and "
727 "positive is up."
728 ),
729 code="scanpath_studio/measures.py:enrich_fixations",
730 output="angle_incoming, angle_outgoing",
731 unit="degrees (−180, 180]",
732 missing="Trial edges ⇒ NaN.",
733 tiers="A, C",
734 status=STATUS_VERIFIED,
735 consumers=(_UI, _API, _EXPORT),
736 tests=("tests/test_measures.py",),
737 ),
738 Computation(
739 id="fix.rebased_onsets",
740 name="Rebased fixation onsets",
741 category=CATEGORY_MEASURE,
742 summary="Trial-relative onset times for animation and time series.",
743 formula=(
744 "Cumulative onsets rebased so the trial starts at 0, from "
745 "`timestamp_ms` where present, else by accumulating durations."
746 ),
747 code="scanpath_studio/measures.py:rebased_fixation_onsets",
748 output="Onset array",
749 unit="ms",
750 missing="A backwards clock restarts the accumulation (see VAL-7).",
751 tiers="A, C",
752 status=STATUS_PARTIAL,
753 consumers=(_UI, _API, _CLI),
754 tests=("tests/test_measures.py",),
755 ),
756 # ------------------------------------------------------------------
757 # Preprocessing
758 # ------------------------------------------------------------------
759 Computation(
760 id="pre.merge_short",
761 name="Short-fixation merging",
762 category=CATEGORY_PREPROCESSING,
763 summary="Fold a short fixation into a neighbour within a character distance.",
764 formula=(
765 "A fixation below the short threshold is merged into the nearer "
766 "adjacent fixation when that neighbour is within the merge distance, "
767 "expressed in characters and converted to px via "
768 "`geom.word_char_advance`. Durations add; position follows the "
769 "survivor."
770 ),
771 precedence=(
772 "#BUG-27: the conversion reads the shared letter scale, so "
773 '"within 1 character" means the same on every word.'
774 ),
775 code="scanpath_studio/preprocessing.py:merge_short_fixations",
776 output="A reduced fixation frame",
777 unit="ms threshold, characters distance",
778 missing="Off by default; original rows stay available.",
779 tiers="A, C",
780 status=STATUS_PARTIAL,
781 reference="A common cleaning step; thresholds are the user's choice.",
782 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT),
783 tests=("tests/test_preprocessing.py",),
784 ),
785 Computation(
786 id="pre.exclude_short",
787 name="Short/long fixation exclusion",
788 category=CATEGORY_PREPROCESSING,
789 summary="Soft-exclude fixations outside a duration window.",
790 formula="Drop fixations shorter than / longer than the chosen bounds.",
791 code="scanpath_studio/preprocessing.py:preprocess_fixations",
792 unit="ms",
793 missing="Soft: excluded rows are reported, not deleted from the source.",
794 tiers="C",
795 status=STATUS_PARTIAL,
796 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT),
797 tests=("tests/test_preprocessing.py",),
798 ),
799 Computation(
800 id="pre.blink_adjacent",
801 name="Blink-adjacent exclusion",
802 category=CATEGORY_PREPROCESSING,
803 summary="Drop fixations immediately before/after a blink.",
804 formula="Exclude the fixations neighbouring any row flagged `is_blink`.",
805 code="scanpath_studio/preprocessing.py:preprocess_fixations",
806 missing="No blink column ⇒ the option has no effect.",
807 tiers="C",
808 status=STATUS_PARTIAL,
809 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT),
810 tests=("tests/test_preprocessing.py",),
811 ),
812 Computation(
813 id="pre.cleaning_report",
814 name="Cleaning QA report",
815 category=CATEGORY_PREPROCESSING,
816 summary="What the preprocessing pass would remove, and why.",
817 formula="Counts per exclusion reason over the unfiltered frame.",
818 code="scanpath_studio/preprocessing.py:cleaning_report",
819 output="Cleaning QA table",
820 tiers="C",
821 status=STATUS_PARTIAL,
822 consumers=(_UI_PREPROCESSING, _API, _CLI, _EXPORT, _INSPECT_DERIVED),
823 tests=("tests/test_preprocessing.py",),
824 ),
825 Computation(
826 id="pre.sentence_measures",
827 name="Sentence-level measures",
828 category=CATEGORY_PREPROCESSING,
829 summary="Per-sentence reading time and counts.",
830 formula=(
831 "Words are grouped into sentences by `infer_sentence_ids` "
832 "(terminal punctuation). Each sentence's durations, fixation and "
833 "run counts, go-past times and skip flag are then derived from the "
834 "fixations on its words; the supplied word measures are not used, "
835 "so a sentence with no fixations reads as skipped (which is why "
836 "Corpus Analysis → Per sentence is held back)."
837 ),
838 code="scanpath_studio/preprocessing.py:sentence_measures",
839 output="Sentences table",
840 unit="ms, counts",
841 missing="Sentence inference is textual, not annotated — approximate.",
842 tiers="C",
843 status=STATUS_PARTIAL,
844 consumers=(_CORPUS, _API, _CLI, _EXPORT, _INSPECT_DERIVED),
845 tests=("tests/test_preprocessing.py",),
846 ),
847 Computation(
848 id="pre.saccade_table",
849 name="Saccade table",
850 category=CATEGORY_PREPROCESSING,
851 summary="One row per saccade, with amplitude, angle and class.",
852 formula=(
853 "Consecutive fixation pairs within a trial; amplitude in px, and in "
854 "degrees only when `pixels_per_degree` is supplied."
855 ),
856 code="scanpath_studio/preprocessing.py:saccade_table",
857 output="Saccades table",
858 unit="px, deg (when geometry is known), ms",
859 missing="Assumed geometry ⇒ the degree columns inherit that assumption.",
860 tiers="C",
861 status=STATUS_PARTIAL,
862 consumers=(_API, _CLI, _EXPORT, _INSPECT_DERIVED),
863 tests=("tests/test_preprocessing.py",),
864 ),
865 Computation(
866 id="pre.character_grid",
867 name="Character grid",
868 category=CATEGORY_PREPROCESSING,
869 summary="Per-character boxes derived from word boxes.",
870 formula=(
871 "Character `k` of a word spans `x + (k−1) × advance` to `x + k × "
872 "advance`, where the advance is `geom.word_char_advance`."
873 ),
874 code="scanpath_studio/preprocessing.py:character_grid",
875 unit="px",
876 missing="Proportional fonts make this an approximation.",
877 precedence=(
878 "#BUG-27: the advance is the shared letter scale, not `width / "
879 "len(text)` — which on a tiling corpus stretched the glyph row "
880 "across the trailing inter-word padding, so each character box after "
881 "the first sat progressively further right than its glyph."
882 ),
883 tiers="A, C",
884 status=STATUS_CONVENTION,
885 consumers=(_API, _CLI, _EXPORT, _INSPECT_DERIVED),
886 tests=("tests/test_preprocessing.py",),
887 ),
888 Computation(
889 id="pre.rtl",
890 name="Right-to-left detection",
891 category=CATEGORY_PREPROCESSING,
892 summary="Whether a word's script runs right to left.",
893 formula="Unicode range test over the word's characters.",
894 code="scanpath_studio/preprocessing.py:detect_right_to_left",
895 output="right_to_left",
896 tiers="A, C",
897 status=STATUS_VERIFIED,
898 consumers=(_UI, _API, _CORPUS),
899 tests=("tests/test_preprocessing.py",),
900 ),
901 Computation(
902 id="pre.sensitivity",
903 name="Measure sensitivity",
904 category=CATEGORY_PREPROCESSING,
905 summary="How much the word measures move under different line assignments.",
906 formula=(
907 "Each trial's fixations are line-assigned by every method in "
908 "`methods` (default `attach`, `slice`, `consensus`), FFD / FPRT / "
909 "RPD / TFD are recomputed per method, and each word's spread (max − "
910 "min across methods) is reported beside a per-trial correction "
911 "report (PRE-18)."
912 ),
913 code="scanpath_studio/preprocessing.py:measure_sensitivity",
914 tiers="C",
915 status=STATUS_PARTIAL,
916 consumers=(_API_EXPERIMENTAL,),
917 tests=("tests/test_preprocessing.py",),
918 ),
919 Computation(
920 id="align.algorithms",
921 name="Vertical drift correction",
922 category=CATEGORY_PREPROCESSING,
923 summary="Line-assignment algorithms, ported natively (PRE-3).",
924 formula=(
925 "The ten Carr et al. algorithms — `attach`, `chain`, `cluster`, "
926 "`compare`, `merge`, `regress`, `segment`, `split`, `stretch`, "
927 "`warp` — plus `slice` and a `consensus` vote over them. Each "
928 "reassigns fixation *y* to a text line. Not in this release "
929 "(#PRE-21)."
930 ),
931 code="scanpath_studio/alignment.py:correct",
932 output="Corrected fixation y (display only; exported tables stay raw)",
933 missing="Off by default; the original coordinates are never overwritten.",
934 tiers="B, C",
935 status=STATUS_PARTIAL,
936 reference=(
937 "Carr, Pescuma, Furlan, Ktori & Crepaldi (2021), *Algorithms for the "
938 "automated correction of vertical drift in eye-tracking data*, "
939 "Behavior Research Methods. Ported from the reference implementation "
940 "— the one entry with a genuine tier-B comparison."
941 ),
942 consumers=(_UI, _API, _CLI),
943 tests=("tests/test_alignment.py", "tests/test_cli_drift.py"),
944 ),
945 # ------------------------------------------------------------------
946 # Aggregation / statistics
947 # ------------------------------------------------------------------
948 Computation(
949 id="agg.measure_values",
950 name="Measure value extraction",
951 category=CATEGORY_AGGREGATION,
952 summary="Pull one registered measure's values out of a frame.",
953 formula=(
954 "The `aggregation.MEASURES` entry names the frame (words or "
955 "fixations), the column and the unit; values are coerced numeric and "
956 "NaNs dropped."
957 ),
958 code="scanpath_studio/aggregation.py:measure_values",
959 missing="Non-numeric entries become NaN and are dropped, not zeroed.",
960 tiers="C, D",
961 status=STATUS_PARTIAL,
962 consumers=(_CORPUS, _API),
963 tests=("tests/test_aggregation.py",),
964 ),
965 Computation(
966 id="agg.aggregate_value",
967 name="Central tendency",
968 category=CATEGORY_AGGREGATION,
969 summary="The Aggregate selector: mean / median / sum.",
970 formula="`np.nanmean` · `np.nanmedian` · `np.nansum` over the values.",
971 code="scanpath_studio/aggregation.py:aggregate_value",
972 missing="NaN-skipping throughout; an all-NaN input gives NaN.",
973 tiers="A, C",
974 status=STATUS_VERIFIED,
975 consumers=(_CORPUS, _API),
976 tests=("tests/test_aggregation.py",),
977 ),
978 Computation(
979 id="agg.spread",
980 name="Spread band",
981 category=CATEGORY_AGGREGATION,
982 summary="The error band drawn around an aggregate.",
983 formula=(
984 "`SD` → ±1 sample std (ddof=1) · `SEM` → ±std/√n · `IQR` → the 25th "
985 "and 75th percentiles · `Bootstrap CI` → `agg.bootstrap_ci`. With "
986 "`agg='sum'`, SD/SEM fall back to the bootstrap: the spread of "
987 "individual observations does not bracket a total."
988 ),
989 code="scanpath_studio/aggregation.py:spread_bounds",
990 missing="Empty input or NaN center ⇒ a zero-width band.",
991 tiers="A, C",
992 status=STATUS_VERIFIED,
993 consumers=(_CORPUS, _API),
994 tests=("tests/test_aggregation.py",),
995 ),
996 Computation(
997 id="agg.bootstrap_ci",
998 name="Bootstrap confidence interval",
999 category=CATEGORY_AGGREGATION,
1000 summary="Percentile bootstrap CI of the chosen aggregate.",
1001 formula=(
1002 "1000 resamples with replacement; the CI is the 2.5th and 97.5th "
1003 "percentiles of the resampled statistic."
1004 ),
1005 code="scanpath_studio/aggregation.py:bootstrap_ci",
1006 unit="same as the measure",
1007 missing="n < 2 ⇒ a degenerate interval at the point estimate.",
1008 precedence="Seeded (`seed=0`) — the same data gives the same interval.",
1009 tiers="A, C",
1010 status=STATUS_VERIFIED,
1011 reference="Percentile bootstrap; no bias correction.",
1012 consumers=(_CORPUS, _API),
1013 tests=("tests/test_aggregation.py",),
1014 ),
1015 Computation(
1016 id="agg.effect_size",
1017 name="Group means and difference",
1018 category=CATEGORY_AGGREGATION,
1019 summary="Two groups' means, their difference and Cohen's d (AN-21).",
1020 formula=(
1021 "Each value is one participant's mean of the measure (pooled "
1022 "observations when the data names no participants). "
1023 "`mean_diff = mean(A) − mean(B)`. Cohen's *d* uses the pooled SD "
1024 "`sqrt(((nA−1)·varA + (nB−1)·varB) / (nA+nB−2))` with ddof=1, and "
1025 "is shown only when the groups share no participant."
1026 ),
1027 code="scanpath_studio/aggregation.py:group_mean_difference",
1028 output="mean_a, mean_b, mean_diff, cohen_d, n_a, n_b",
1029 grouping="One value per participant in each group",
1030 missing=(
1031 "n < 2 in either group ⇒ NaN *d*. A zero pooled SD gives "
1032 "**NaN**, not 0.0, so it cannot read as 'no effect' beside a "
1033 "non-zero mean difference."
1034 ),
1035 tiers="A, C",
1036 status=STATUS_PARTIAL,
1037 reference=(
1038 "**Descriptive only** — no significance test. A participant in both "
1039 "groups contributes to both means, so the groups are not "
1040 "independent samples."
1041 ),
1042 consumers=(_CORPUS, _API),
1043 tests=("tests/test_aggregation.py",),
1044 ),
1045 Computation(
1046 id="agg.group_mask",
1047 name="Group definition",
1048 category=CATEGORY_AGGREGATION,
1049 summary="Which rows belong to a cohort.",
1050 formula=(
1051 "A spec maps column → allowed values; the mask is the conjunction of "
1052 "membership tests. Two modes: split one field, or two independent "
1053 "filter sets. A key may be a tuple of columns matched as one "
1054 "composite key: a trial-metadata field resolves to the "
1055 "(participant, trial) readings its rows describe, a participant field to "
1056 "participant ids and a text field to text ids — the tables are never "
1057 "joined onto the frames."
1058 ),
1059 code="scanpath_studio/aggregation.py:group_mask",
1060 missing=(
1061 "A column (or any column of a composite key) absent from the frame "
1062 "contributes no constraint; a metadata selection that matches "
1063 "nothing selects no rows."
1064 ),
1065 tiers="A, C",
1066 status=STATUS_VERIFIED,
1067 consumers=(_CORPUS, _API),
1068 tests=("tests/test_aggregation.py",),
1069 ),
1070 Computation(
1071 id="agg.word_profile",
1072 name="Per-word cohort profile",
1073 category=CATEGORY_AGGREGATION,
1074 summary="A measure per word position, aggregated across participants.",
1075 formula="Group the word measures by word id and apply `agg.aggregate_value`.",
1076 code="scanpath_studio/aggregation.py:cohort_word_profile",
1077 missing="A minimum-participants threshold drops thinly-sampled words.",
1078 tiers="C",
1079 status=STATUS_PARTIAL,
1080 consumers=(_CORPUS, _API),
1081 tests=("tests/test_aggregation.py",),
1082 ),
1083 Computation(
1084 id="agg.word_rates",
1085 name="Skip / regression rate profile",
1086 category=CATEGORY_AGGREGATION,
1087 summary="Rate measures per word.",
1088 formula=(
1089 "Mean of the 0/1 flag over the participants who reported it — a "
1090 "proportion in [0, 1]. Each rate has its own participant count "
1091 "(`n_skip`, `n_regression_in`) and its own minimum-participants verdict."
1092 ),
1093 code="scanpath_studio/aggregation.py:word_rate_profile",
1094 unit="proportion",
1095 missing=(
1096 "A missing flag is no observation: it is left out of that rate and "
1097 "its participant count, never read as 0. A rate below the minimum "
1098 "participants is hidden; the word stays while its other rate stands."
1099 ),
1100 tiers="A, C",
1101 status=STATUS_PARTIAL,
1102 consumers=(_CORPUS, _API),
1103 tests=("tests/test_aggregation.py",),
1104 ),
1105 Computation(
1106 id="agg.reader_summary",
1107 name="Per-participant summary",
1108 category=CATEGORY_AGGREGATION,
1109 summary="One row per participant: totals, means and rates.",
1110 formula=(
1111 "Counts and NaN-skipping means over that participant's rows. "
1112 "`mean_saccade_px` is the mean of `fix.saccade_amplitude` and is "
1113 "in pixels."
1114 ),
1115 code="scanpath_studio/aggregation.py:reader_summary_table",
1116 output="Readers table",
1117 unit="ms, px, counts, proportions",
1118 tiers="C, D",
1119 status=STATUS_PARTIAL,
1120 consumers=(_CORPUS, _EXPORT, _INSPECT, _API),
1121 tests=("tests/test_aggregation.py",),
1122 ),
1123 Computation(
1124 id="agg.trial_summary",
1125 name="Per-trial summary",
1126 category=CATEGORY_AGGREGATION,
1127 summary="One row per trial: reading time, counts, rates.",
1128 formula=(
1129 "Counts and sums over the trial's fixations and word measures. "
1130 "`reading_time_ms` is last fixation end − first fixation start; "
1131 "without recorded fixation onsets it is the summed fixation "
1132 "durations, and `reading_time_source` says it is an estimate. "
1133 "`wpm` = words ÷ reading time."
1134 ),
1135 missing=(
1136 "No onset column ⇒ reading time and wpm are duration-based "
1137 "estimates, labeled as such — never the 0, 1, 2, … order numbers."
1138 ),
1139 code="scanpath_studio/aggregation.py:trial_summary_table",
1140 output="Trials table",
1141 unit="ms, counts",
1142 tiers="C, D",
1143 status=STATUS_PARTIAL,
1144 consumers=(_CORPUS, _EXPORT, _INSPECT, _API),
1145 tests=("tests/test_aggregation.py",),
1146 ),
1147 Computation(
1148 id="agg.normalize",
1149 name="Normalized measure column",
1150 category=CATEGORY_AGGREGATION,
1151 summary="Rescale a measure for cross-participant comparison.",
1152 formula=(
1153 "Per-participant z-score, `(value − participant mean) / participant SD`, "
1154 "when **Z-score per participant** is on."
1155 ),
1156 code="scanpath_studio/aggregation.py:add_normalized_column",
1157 missing=(
1158 "A participant with zero variance (or one value) ⇒ 0, the participant's own "
1159 "mean; a missing value stays NaN."
1160 ),
1161 tiers="A, C",
1162 status=STATUS_PARTIAL,
1163 consumers=(_CORPUS,),
1164 tests=("tests/test_aggregation.py",),
1165 ),
1166 Computation(
1167 id="agg.landing_curve",
1168 name="Landing-position curve",
1169 category=CATEGORY_AGGREGATION,
1170 summary="Distribution of initial landing positions by word length.",
1171 formula=(
1172 "Histogram of the landing position as a *fraction of the word's "
1173 "interest area* — `(first_fix_x − word.x) / width` over the "
1174 "experiment's own box, i.e. `(measure.landing_position − 1)` over "
1175 "the box's `width / geom.word_char_advance` character cells (RTL "
1176 "counted from where the glyphs end, as the letter position is). "
1177 "Unclipped — binned per word length."
1178 ),
1179 code="scanpath_studio/aggregation.py:landing_positions",
1180 unit=(
1181 "fraction of the interest area (0–1 for a landing inside the box), "
1182 "or px with `as_fraction=False`"
1183 ),
1184 precedence=(
1185 "#BUG-83: on a glyph-tight corpus the box is the glyph run, so 0 is "
1186 "the first letter's edge and 1 the last's. On a tiling corpus the "
1187 "box's last cell is the space after the word, so the glyphs fill "
1188 "`[0, n / (n + 1))` and a landing on that space reads just below 1, "
1189 "not clipped onto 1.0. A first fixation assigned from outside the box "
1190 "(by an imported `word_id`) reads below 0 or above 1 rather than "
1191 "being clipped onto an edge. The origin is the word's `x` and the "
1192 "scale is `geom.word_char_advance`."
1193 ),
1194 tiers="C",
1195 status=STATUS_PARTIAL,
1196 consumers=(_CORPUS,),
1197 tests=("tests/test_aggregation.py",),
1198 ),
1199 Computation(
1200 id="agg.over_time",
1201 name="Trend over time",
1202 category=CATEGORY_AGGREGATION,
1203 summary="A measure by trial index or fixation index.",
1204 formula="Aggregate per index position across the selection.",
1205 code="scanpath_studio/aggregation.py:metric_over_time",
1206 missing="Index positions with no data are gaps, not zeros.",
1207 tiers="C",
1208 status=STATUS_PARTIAL,
1209 consumers=(_CORPUS,),
1210 tests=("tests/test_aggregation.py",),
1211 ),
1212 # ------------------------------------------------------------------
1213 # Similarity
1214 # ------------------------------------------------------------------
1215 Computation(
1216 id="sim.nld",
1217 name="Normalized Levenshtein distance",
1218 category=CATEGORY_SIMILARITY,
1219 summary="Scanpath similarity over AoI sequences.",
1220 formula=(
1221 "`levenshtein(a, b) / max(len(a), len(b))` ∈ [0, 1]; 0 is identical. "
1222 "Two empty sequences give 0."
1223 ),
1224 code="scanpath_studio/similarity.py:normalized_levenshtein",
1225 unit="dimensionless (0–1)",
1226 missing="Not in this release.",
1227 tiers="A, C",
1228 status=STATUS_VERIFIED,
1229 reference="Standard edit-distance scanpath comparison.",
1230 consumers=(_UI, _API),
1231 tests=("tests/test_similarity.py",),
1232 ),
1233 Computation(
1234 id="sim.aoi_sequence",
1235 name="AoI sequence",
1236 category=CATEGORY_SIMILARITY,
1237 summary="The symbol string an NLD comparison runs on.",
1238 formula=(
1239 "Assigned `word_id`s in fixation order, with unassigned fixations "
1240 "dropped and (optionally) immediate repeats collapsed."
1241 ),
1242 code="scanpath_studio/similarity.py:aoi_sequence",
1243 missing="A trial with no assigned fixations yields an empty sequence.",
1244 tiers="A, C",
1245 status=STATUS_VERIFIED,
1246 consumers=(_UI, _API),
1247 tests=("tests/test_similarity.py",),
1248 ),
1249 Computation(
1250 id="sim.windowed",
1251 name="NLD by fixation index / time",
1252 category=CATEGORY_SIMILARITY,
1253 summary="Similarity restricted to a window of the scanpath.",
1254 formula="`sim.nld` over the sub-sequence inside the index or time window.",
1255 code="scanpath_studio/similarity.py:nld_by_fixation_index",
1256 tiers="C",
1257 status=STATUS_PARTIAL,
1258 consumers=(_UI, _API),
1259 tests=("tests/test_similarity.py",),
1260 ),
1261 # ------------------------------------------------------------------
1262 # Unit / coordinate conversion
1263 # ------------------------------------------------------------------
1264 Computation(
1265 id="geom.pixels_per_degree",
1266 name="Pixels per degree of visual angle",
1267 category=CATEGORY_GEOMETRY,
1268 summary="The screen-geometry conversion every angular unit depends on.",
1269 formula=(
1270 "`px_per_mm = canvas_width_px / monitor_width_mm`; "
1271 "`mm_per_degree = 2 · viewing_distance_mm · tan(0.5°)`; "
1272 "`px_per_degree = px_per_mm · mm_per_degree`."
1273 ),
1274 code="scanpath_studio/experimental_setup.py:pixels_per_degree",
1275 unit="px / degree",
1276 missing="Any missing geometry ⇒ no conversion is offered at all.",
1277 precedence=(
1278 "**Provenance matters more than the number.** Every built-in corpus "
1279 "assumes its monitor size and viewing distance, so a degree-valued "
1280 "result inherits that — see the *Recording setup* panel."
1281 ),
1282 tiers="A, C",
1283 status=STATUS_VERIFIED,
1284 consumers=(_UI, _API, _CLI, _EXPORT),
1285 tests=("tests/test_experimental_setup.py",),
1286 ),
1287 Computation(
1288 id="geom.font_pt_to_px",
1289 name="Font point size to pixels",
1290 category=CATEGORY_GEOMETRY,
1291 summary="Typography conversion for true-scale text rendering.",
1292 formula="`px = pt · dpi / 72`.",
1293 code="scanpath_studio/experimental_setup.py:font_pt_to_px",
1294 unit="px",
1295 tiers="A, C",
1296 status=STATUS_VERIFIED,
1297 consumers=(_UI, _API, _CLI),
1298 tests=("tests/test_experimental_setup.py",),
1299 ),
1300 Computation(
1301 id="geom.word_box_bounds",
1302 name="Word interest-area edges",
1303 category=CATEGORY_GEOMETRY,
1304 summary="Where one word's interest area ends and the next begins.",
1305 formula=(
1306 "`x .. x + width` by `y .. y + height` — the experiment's own "
1307 "rectangles, unmodified. On a tiling corpus each box includes the "
1308 "space after its word."
1309 ),
1310 code="scanpath_studio/measures.py:word_box_bounds",
1311 unit="px",
1312 precedence=(
1313 "The boundary *between* words, for everything that tests a point "
1314 "against a box or draws one: `assign.fixation_to_word`, "
1315 "`assign.in_text`, the drawn outlines, the word heatmaps, the "
1316 "critical-span frame, drift correction and the model scanpaths. A "
1317 "position *inside* a word goes through `geom.word_char_advance` "
1318 "instead, and where its letters are through `geom.word_glyph_span`; the drawn word label is centered in the box (#BUG-97)."
1319 ),
1320 tiers="A, C",
1321 status=STATUS_PARTIAL,
1322 consumers=(_UI, _API, _CORPUS),
1323 tests=("tests/test_word_box_geometry.py", "tests/test_word_id_offset.py"),
1324 ),
1325 Computation(
1326 id="geom.word_box_space_px",
1327 name="Inter-word padding baked into each box",
1328 category=CATEGORY_GEOMETRY,
1329 summary="Detects a tiling layout that carries one trailing space per box.",
1330 formula=(
1331 "Median of `width / (len(text) + 1)` across one trial's words — the "
1332 "advance — reported only when the boxes are consistently that wide "
1333 "**and** actually tile (no gaps). Anything else ⇒ `0.0`, i.e. "
1334 "'these AOIs are glyph-tight — each box is its glyph run'."
1335 ),
1336 code="scanpath_studio/measures.py:word_box_space_px",
1337 unit="px",
1338 missing="No usable words ⇒ 0.0 (glyph-tight), never a guess.",
1339 precedence=(
1340 "Never moves a box edge (#BUG-83); it only tells "
1341 "`geom.word_char_advance` and `geom.word_glyph_span` how many "
1342 "character cells a box holds."
1343 ),
1344 tiers="A, C",
1345 status=STATUS_VERIFIED,
1346 consumers=(_UI, _API, _EXPORT),
1347 tests=("tests/test_measures.py",),
1348 ),
1349 Computation(
1350 id="geom.word_char_advance",
1351 name="Character advance within a word",
1352 category=CATEGORY_GEOMETRY,
1353 summary="How wide one letter is — the scale for every within-word position.",
1354 formula=(
1355 "`width / (len(text) + 1)` when `geom.word_box_space_px` finds "
1356 "trailing padding, else `width / len(text)`."
1357 ),
1358 code="scanpath_studio/measures.py:word_char_advance",
1359 unit="px / character",
1360 missing="No `text`/`width` ⇒ NaN, and the letter measures report NaN.",
1361 precedence=(
1362 "The single accessor for the letter scale, as `geom.word_box_bounds` "
1363 "is for the boundary between words: `measure.landing_position`, "
1364 "`measure.landing_distance`, `agg.landing_curve` and the saccade "
1365 "table's launch/landing letter all read it. #BUG-27 — before that "
1366 "each derived its own `width / len(text)`, which is one advance too "
1367 "wide on a tiling corpus, by a factor that varied with word length."
1368 ),
1369 tiers="A, C",
1370 status=STATUS_VERIFIED,
1371 consumers=(_UI, _API, _EXPORT, _CORPUS),
1372 tests=("tests/test_measures.py",),
1373 ),
1374 Computation(
1375 id="geom.word_glyph_span",
1376 name="Where a word's glyphs are",
1377 category=CATEGORY_GEOMETRY,
1378 summary="The glyph run inside a word's box — where its letters are.",
1379 formula=(
1380 "Starts at `x` and runs `len(text) × geom.word_char_advance`: the "
1381 "whole box on a glyph-tight corpus, one advance short of it on a "
1382 "tiling one. No `text` ⇒ the box width."
1383 ),
1384 code="scanpath_studio/measures.py:word_glyph_span",
1385 unit="px",
1386 precedence=(
1387 "Not an interest area: `agg.landing_curve` measures a landing "
1388 "across it and mirrors an RTL one. Measured against "
1389 "OneStop's own Experiment Builder screens: each tiling box is "
1390 "centered on its word, half a space either side, so the run's "
1391 "`x` start is half an advance early there; the label and snap "
1392 "use the box center, the landing measures do not."
1393 ),
1394 tiers="A",
1395 status=STATUS_PARTIAL,
1396 consumers=(_UI, _API, _CORPUS),
1397 tests=("tests/test_word_box_geometry.py",),
1398 ),
1399 # ------------------------------------------------------------------
1400 # Display / export transformations
1401 # ------------------------------------------------------------------
1402 Computation(
1403 id="disp.marker_sizes",
1404 name="Fixation marker sizing",
1405 category=CATEGORY_DISPLAY,
1406 summary="Marker size encodes fixation duration on one fixed scale.",
1407 formula=(
1408 "Fixed scales (`marker_size_scale` = `sqrt`, the default; `linear`; "
1409 "`log`): `size = s_min + (s_max − s_min) · (f(d) − f(lo)) / "
1410 "(f(hi) − f(lo))`, with `d` clamped to the duration bounds "
1411 "`[lo, hi]` (`marker_duration_range`, default 50–600 ms) and `f` "
1412 "= √, identity or ln. `relative`: linear between the drawn set's "
1413 "own shortest and longest duration (the scale before the fixed one; "
1414 "older saved configs and Share links keep it). **Display only** — "
1415 "never a recorded value."
1416 ),
1417 code="scanpath_studio/plots.py:_compute_marker_sizes",
1418 unit="px (marker diameter)",
1419 missing="A missing duration is treated as 0 ms: the smallest marker.",
1420 precedence=(
1421 "One scale for single-trial figures, both comparison sides, replays "
1422 "and bulk exports, so a duration draws at one size in all of them; "
1423 "only the px range is per scanpath in Compare."
1424 ),
1425 tiers="C, D",
1426 status=STATUS_CONVENTION,
1427 consumers=(_UI, _API, _CLI, _EXPORT),
1428 tests=(
1429 "tests/test_duration_scale.py",
1430 "tests/test_plots.py",
1431 "tests/test_builder_parity.py",
1432 ),
1433 ),
1434 Computation(
1435 id="disp.axis_ranges",
1436 name="Axis ranges and inversion",
1437 category=CATEGORY_DISPLAY,
1438 summary="Screen coordinates, drawn the way the screen is.",
1439 formula=(
1440 "The y axis is inverted (`y_range = [max, min]`) so the figure "
1441 "matches the display; ranges come from the canvas, not the data, "
1442 "when a canvas size is known."
1443 ),
1444 code="scanpath_studio/plots.py:_compute_axis_ranges",
1445 unit="px",
1446 tiers="C, D",
1447 status=STATUS_CONVENTION,
1448 consumers=(_UI, _API, _CLI, _EXPORT),
1449 tests=("tests/test_plots.py",),
1450 ),
1451 Computation(
1452 id="disp.true_scale",
1453 name="True-scale text rendering",
1454 category=CATEGORY_DISPLAY,
1455 summary="One line of text fills its share of the recorded line pitch.",
1456 formula=(
1457 "A word label's font is `1/line_spacing` of the line pitch (the median "
1458 "line-to-line distance of the word boxes), capped so the words fit "
1459 "their box widths (`plots._width_fit_font`; the smaller wins), in "
1460 "data pixels converted at the figure's display scale. When the "
1461 "boxes are monospace words padded alike — half the gap to each "
1462 "neighbour — the font is read off them instead: one character cell "
1463 "is the slope of box width over word length "
1464 "(`plots._padded_monospace_font`), over the font's advance. The "
1465 "figure is drawn at its exact pixel size and scaled as one block."
1466 ),
1467 code="scanpath_studio/tabs.py:_render_true_scale_chart",
1468 # The spatial plot must stay on this path: `st.plotly_chart` loses the
1469 # scale guarantee (a developer rule, so not published).
1470 tiers="D",
1471 status=STATUS_CONVENTION,
1472 consumers=(_UI,),
1473 tests=("tests/test_plots.py",),
1474 ),
1475 Computation(
1476 id="disp.animation_timing",
1477 name="Animation timing",
1478 category=CATEGORY_DISPLAY,
1479 summary="How recorded time maps to playback time.",
1480 formula=(
1481 "Frames sit on a uniform reading-time grid over `fix.rebased_onsets`; "
1482 "the frame at reading time t is on screen once t / playback speed of "
1483 "wall time has passed, so a replay lasts reading span / speed. The "
1484 "player keeps that clock itself, skipping frames a display is too "
1485 "slow to show, and a GIF/MP4 lasts the same. A multipart replay "
1486 "changes screen at the boundary and draws no connector across "
1487 "canvases."
1488 ),
1489 code="scanpath_studio/plots.py:make_scanpath_animation",
1490 unit="ms (recorded) → ms (playback)",
1491 precedence=(
1492 "Plotly's own frame queue is never the clock: it rounds every frame "
1493 "up to whole display ticks and the error accumulates (BUG-93). "
1494 "Without the player (`fig.show()`) the figure falls back to it."
1495 ),
1496 tiers="C, D",
1497 status=STATUS_CONVENTION,
1498 consumers=(_UI, _API, _CLI, _EXPORT),
1499 tests=("tests/test_replay_player.py", "tests/test_animation_export.py"),
1500 ),
1501 Computation(
1502 id="disp.illustration",
1503 name="Illustration disclosure",
1504 category=CATEGORY_DISPLAY,
1505 summary="When a figure stops being a faithful record.",
1506 formula=(
1507 "Views that no longer show the data as recorded — snapped "
1508 "fixations, arced saccades, hidden or windowed fixations, a replay "
1509 "not at real time, an authored scanpath — are labeled *Illustration*."
1510 ),
1511 code="scanpath_studio/illustration.py:illustration_reasons",
1512 tiers="C, D",
1513 status=STATUS_VERIFIED,
1514 consumers=(_UI, _API, _CLI, _EXPORT),
1515 tests=("tests/test_illustration.py", "tests/test_disclosure.py"),
1516 ),
1517)
1519BY_ID = {entry.id: entry for entry in REGISTER}
1521#: Entries the default build does not compute: they need
1522#: ``SCANPATH_EXPERIMENTAL=1`` (`constants.computed_measures_enabled`,
1523#: `preprocessing_enabled`, `drift_correction_enabled`, `similarity_enabled`).
1524#: The page marks them, so it does not present held-back work as shipped.
1525EXPERIMENTAL_IDS = frozenset(
1526 {
1527 "measure.ffd",
1528 "measure.fprt",
1529 "measure.rpd",
1530 "measure.tfd",
1531 "measure.nfix",
1532 "measure.skip",
1533 "measure.regressions",
1534 "measure.landing_position",
1535 "measure.landing_distance",
1536 "measure.second_pass",
1537 "measure.single_fix",
1538 "measure.reg_in_count",
1539 "pre.merge_short",
1540 "pre.exclude_short",
1541 "pre.blink_adjacent",
1542 "pre.cleaning_report",
1543 "pre.sentence_measures",
1544 "pre.saccade_table",
1545 "pre.character_grid",
1546 "pre.sensitivity",
1547 "align.algorithms",
1548 "agg.reader_summary",
1549 "agg.trial_summary",
1550 "agg.landing_curve",
1551 "sim.nld",
1552 "sim.aoi_sequence",
1553 "sim.windowed",
1554 }
1555)
1557_TRACKER_ID = r"#?[A-Z]{2,5}-\d+"
1558_TRACKER_PATTERNS = (
1559 # "(BUG-25)", "(#PRE-21)", "(PRE-11/12)", "(BUG-61..66)", "(see VIZ-8)"
1560 re.compile(
1561 rf"\s*\((?:see |cf\. )?{_TRACKER_ID}"
1562 rf"(?:\s*(?:[,/]|\.\.|and|&)\s*(?:{_TRACKER_ID}|\d+))*\)"
1563 ),
1564 # "only with SCANPATH_EXPERIMENTAL=1 — PRE-22)", "blanked, BUG-63)"
1565 re.compile(rf"\s*[—–,-]\s*{_TRACKER_ID}(?=\))"),
1566 # a sentence opening "#BUG-27: " or "#BUG-27 — "
1567 re.compile(rf"(?:^|(?<=\. )){_TRACKER_ID}(?::| —)\s+"),
1568)
1571def _public(text: str) -> str:
1572 """The register's text as the published page shows it: tracker ids are
1573 for the code and its history, not for a reader of the docs."""
1574 *inline, opening = _TRACKER_PATTERNS
1575 for pattern in inline:
1576 text = pattern.sub("", text)
1577 # Only a sentence whose opening id was dropped gets a capital again —
1578 # column names (`trial_id`) and formulas (`x .. x + width`) keep their case.
1579 text = opening.sub("\0", text)
1580 return re.sub("\0([a-z]?)", lambda m: m.group(1).upper(), text)
1583def _experimental_note(entry: Computation) -> list[str]:
1584 if entry.id not in EXPERIMENTAL_IDS:
1585 return []
1586 if entry.category == CATEGORY_MEASURE:
1587 body = (
1588 "Scanpath Studio does not compute this in this release. A value "
1589 "your dataset brings is shown as given, defined by the software "
1590 "that exported it."
1591 )
1592 else:
1593 body = "Not in this release."
1594 return ['!!! warning "Experimental"', "", f" {body}", ""]
1597def entries_in(category: str) -> tuple[Computation, ...]:
1598 """Every register entry in one category, in declaration order."""
1599 return tuple(entry for entry in REGISTER if entry.category == category)
1602def measure_entry(column: str) -> Computation | None:
1603 """The reading-measure or fixation entry that defines ``column`` — the one
1604 whose ``output`` names it — or ``None`` for a value the register does not
1605 derive (a fixation's recorded duration). Lets the app quote the register's
1606 own summary and unit beside a measure instead of keeping a second copy."""
1607 for entry in REGISTER:
1608 if not entry.id.startswith(("measure.", "fix.")):
1609 continue
1610 if column in (part.strip() for part in entry.output.split(",")):
1611 return entry
1612 return None
1615def anchor(entry_id: str) -> str:
1616 """The page anchor of one entry — its id, so a link to ``measure.ffd``
1617 survives any rewording of the entry's name (``#measure-ffd``)."""
1618 return entry_id.replace(".", "-").replace("_", "-")
1621def to_markdown() -> str:
1622 """Render the register as the ``docs/computations.md`` page.
1624 Generated rather than hand-maintained so the documentation cannot drift
1625 from the catalogue the integrity tests check.
1626 """
1627 lines: list[str] = [
1628 "<!-- Generated by `python -m scanpath_studio.computations`. Do not edit. -->",
1629 "",
1630 "# Computations & methodology",
1631 "",
1632 f"Register version **{REGISTER_VERSION}** · "
1633 f"{len(REGISTER)} entries across {len(CATEGORIES)} categories.",
1634 "",
1635 "Every operation that derives or semantically changes a value you can "
1636 "see, export, or fetch through the API is listed here with its formula, "
1637 "its units, and how far it has actually been verified. Pure layout and "
1638 "byte-preserving file I/O are out of scope; filtering, precedence and "
1639 "assignment are in, because they change *which observations* a result "
1640 "stands for.",
1641 "",
1642 "## How to read the status column",
1643 "",
1644 "| Status | Means |",
1645 "| --- | --- |",
1646 "| **Verified** | A hand-calculated oracle or exact invariant exists and passes. |",
1647 "| **Partially verified** | Tested, but without an independent reference implementation. |",
1648 "| **Unverified** | Exercised by tests only for execution, not for meaning. |",
1649 "| **Intentional convention** | A choice that can only be documented, not proved. |",
1650 "",
1651 "Verification tiers: **A** hand-calculated synthetic oracle · **B** "
1652 "independent reference implementation · **C** property/invariant tests · "
1653 "**D** cross-surface parity (UI, API, CLI, export agree).",
1654 "",
1655 '!!! note "Tier B is largely absent, on purpose"',
1656 "",
1657 " Comparing against an independent implementation "
1658 "[is planned](https://github.com/lacclab/scanpath-studio/issues/130). "
1659 "Scientific measures therefore read *Partially verified* even where "
1660 "their hand oracle is exact.",
1661 "",
1662 "Entries marked *experimental* are not in this release. They are "
1663 "listed so that their definitions are on record.",
1664 "",
1665 "## Summary",
1666 "",
1667 "| ID | Name | Category | Unit | Status |",
1668 "| --- | --- | --- | --- | --- |",
1669 ]
1670 for entry in REGISTER:
1671 status = entry.status + (
1672 " · experimental" if entry.id in EXPERIMENTAL_IDS else ""
1673 )
1674 lines.append(
1675 f"| [`{entry.id}`](#{anchor(entry.id)}) | {entry.name} | "
1676 f"{entry.category} | {entry.unit or '—'} | {status} |"
1677 )
1678 lines.append("")
1679 for category in CATEGORIES:
1680 entries = entries_in(category)
1681 if not entries:
1682 continue
1683 lines += [f"## {category}", ""]
1684 for entry in entries:
1685 lines += [
1686 f"### `{entry.id}` — {entry.name} {{ #{anchor(entry.id)} }}",
1687 "",
1688 *_experimental_note(entry),
1689 _public(entry.summary),
1690 "",
1691 f"**Formula.** {_public(entry.formula)}",
1692 "",
1693 ]
1694 rows = [
1695 ("Output", _public(entry.output)),
1696 ("Unit", entry.unit),
1697 ("Grouping / ordering", _public(entry.grouping)),
1698 ("Missing & edge cases", _public(entry.missing)),
1699 ("Precedence & caveats", _public(entry.precedence)),
1700 ("Reference", _public(entry.reference)),
1701 ("Code", f"`{entry.code}`"),
1702 ("Consumers", ", ".join(_public(c) for c in entry.consumers)),
1703 ("Tests", ", ".join(f"`{t}`" for t in entry.tests)),
1704 ("Verification", f"tier {entry.tiers} — **{entry.status}**"),
1705 ]
1706 lines += ["| | |", "| --- | --- |"]
1707 for label, value in rows:
1708 if value:
1709 lines.append(f"| **{label}** | {value} |")
1710 lines.append("")
1711 return "\n".join(lines).rstrip() + "\n"
1714def main() -> None: # pragma: no cover - thin CLI wrapper
1715 from pathlib import Path
1717 target = Path(__file__).resolve().parents[1] / "docs" / "computations.md"
1718 target.write_text(to_markdown(), encoding="utf-8")
1719 print(f"Wrote {target} ({len(REGISTER)} entries)")
1722if __name__ == "__main__": # pragma: no cover
1723 main()