Coverage for scanpath_studio/analysis_recipe.py: 88%
78 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""AN-34: the analysis recipe — how one Corpus Analysis table was made.
3A small JSON file downloaded beside each Corpus Analysis table. It records what
4the table cannot say about itself: the app version, the dataset it was read
5from, the trial filters that shaped the pool, the analysis choices (text,
6screen, measure, aggregation, normalization, spread, minimum readers, group
7definitions) and the counts behind the result.
9It **references** the dataset by name and never embeds it: no table rows, and
10no annotation notes (a *Favorites only* or tag filter is recorded as the filter
11it is, not as the annotations behind it). Filters and figure settings are kept
12apart on purpose — this file holds the filters and no styling; the figure
13settings file (🔗 Share → File) holds styling and no filters — so the recipe
14says so in ``excludes`` rather than leaving a reader to guess.
16It describes the current analysis only. There is no runner and no workspace
17format: nothing reads a recipe back.
19Pure — no Streamlit; ``tabs._download_tidy`` assembles the inputs.
20"""
22from __future__ import annotations
24from collections.abc import Mapping, Sequence
25from typing import Any
27import numpy as np
28import pandas as pd
30#: What a recipe is, so a script can tell it from the figure settings file.
31RECIPE_KIND = "scanpath-studio/analysis-recipe"
32#: Bumped when a field changes meaning; adding a field does not bump it.
33RECIPE_VERSION = 1
35#: What the recipe deliberately leaves out, written into every one.
36EXCLUDES = {
37 "figure_settings": (
38 "Palette, canvas and fonts live in the settings file (Scanpath → "
39 "Share → File), which holds no trial filters."
40 ),
41 "data": (
42 "No table rows and no annotation notes. The dataset is referenced by "
43 "name and must be loaded to repeat the analysis."
44 ),
45}
48def jsonable(value: Any) -> Any:
49 """``value`` with sets, tuples and numpy scalars made JSON-native.
51 Sets are sorted (by their text) so the same analysis writes the same file.
52 """
53 if isinstance(value, Mapping):
54 return {str(k): jsonable(v) for k, v in value.items()}
55 if isinstance(value, (set, frozenset)):
56 return [jsonable(v) for v in sorted(value, key=str)]
57 if isinstance(value, (list, tuple)):
58 return [jsonable(v) for v in value]
59 if isinstance(value, (np.ndarray, pd.Index, pd.Series)):
60 return [jsonable(v) for v in value.tolist()]
61 if isinstance(value, np.bool_):
62 return bool(value)
63 if isinstance(value, np.integer):
64 return int(value)
65 if isinstance(value, np.floating):
66 return None if np.isnan(value) else float(value)
67 if isinstance(value, float) and np.isnan(value):
68 return None
69 if value is None or isinstance(value, (str, int, float, bool)):
70 return value
71 return str(value)
74def group_definition(label: str, spec: Mapping | None) -> dict:
75 """One cohort as ``{"label", "constraints"}`` — the spec the views used.
77 ``spec`` is ``aggregation.group_mask``'s ``{column: allowed values}``; a
78 composite key (a trial-metadata cohort's ``(participant_id, trial_id)``)
79 keeps its columns as a list. An empty spec is the whole pool.
80 """
81 constraints = [
82 {
83 "field": list(col) if isinstance(col, tuple) else str(col),
84 "values": jsonable(values),
85 }
86 for col, values in (spec or {}).items()
87 if values is not None
88 ]
89 return {"label": str(label), "constraints": constraints}
92def analysis_choices(
93 *,
94 section: str | None = None,
95 view: str | None = None,
96 text: tuple[str, Any] | None = None,
97 screen: Any = None,
98 reader: Any = None,
99 measure: Any = None,
100 measures: Sequence[Any] | None = None,
101 aggregation: str | None = None,
102 normalize: bool | None = None,
103 spread: str | None = None,
104 min_readers: int | None = None,
105 feature: str | None = None,
106 x_axis: str | None = None,
107 groups: Sequence[dict] | None = None,
108) -> dict:
109 """The choices that made one table, with what does not apply left out.
111 ``measure`` is an ``aggregation.Measure`` (anything with ``key`` / ``label``
112 / ``is_rate``). ``normalize`` is recorded as it was *applied*: the
113 aggregation helpers never z-score a 0–1 rate, so a rate reads ``none``
114 whatever the toggle says.
115 """
116 out: dict = {}
117 if section:
118 out["section"] = section
119 if view:
120 out["view"] = view
121 if text is not None:
122 out["text"] = {"field": str(text[0]), "id": jsonable(text[1])}
123 if screen is not None:
124 out["screen"] = jsonable(screen)
125 if reader is not None:
126 out["reader"] = jsonable(reader)
127 if measure is not None:
128 out["measure"] = {"key": measure.key, "label": measure.label}
129 if measures:
130 out["measures"] = [{"key": m.key, "label": m.label} for m in measures]
131 if aggregation:
132 out["aggregation"] = aggregation
133 if normalize is not None:
134 rate = bool(getattr(measure, "is_rate", False))
135 out["normalization"] = (
136 "z-score per participant" if normalize and not rate else "none"
137 )
138 if spread:
139 out["spread"] = spread
140 if min_readers is not None:
141 out["min_readers"] = int(min_readers)
142 if feature:
143 out["feature"] = feature
144 if x_axis:
145 out["x_axis"] = x_axis
146 if groups:
147 out["groups"] = list(groups)
148 return out
151def result_counts(table: pd.DataFrame | None, extra: Mapping | None = None) -> dict:
152 """What the table itself says about its size: rows, and readers when it
153 names them. ``extra`` adds counts the view already computed (a cohort's
154 readers and fixations), never recomputed here."""
155 out: dict = {"rows": 0 if table is None else len(table)}
156 if table is not None and "participant_id" in table.columns:
157 out["readers"] = int(table["participant_id"].astype(str).nunique())
158 if extra:
159 out.update(jsonable(dict(extra)))
160 return out
163def build_analysis_recipe(
164 *,
165 app_version: str,
166 dataset: Mapping,
167 trial_filters: Sequence[Mapping],
168 pool: Mapping,
169 analysis: Mapping,
170 table_file: str,
171 counts: Mapping,
172 exported_at: str | None = None,
173) -> dict:
174 """The recipe for one Corpus Analysis table.
176 ``trial_filters`` is ``controls.active_filter_items`` — one
177 ``{"field", "values" | "range"}`` entry per filter narrowing the pool (a
178 range also carries ``"unknown": "kept" | "excluded"`` — what it does with
179 the records that have no value); an
180 empty list is an unfiltered pool. ``pool`` holds its trial and reader
181 counts against the dataset's.
182 """
183 recipe = {
184 "kind": RECIPE_KIND,
185 "version": RECIPE_VERSION,
186 "app": {"name": "Scanpath Studio", "version": str(app_version)},
187 "exported_at": exported_at,
188 "dataset": jsonable(dict(dataset)),
189 "trial_filters": [
190 {
191 k: jsonable(v)
192 for k, v in item.items()
193 if k in ("field", "values", "range", "unknown")
194 }
195 for item in trial_filters
196 ],
197 "pool": jsonable(dict(pool)),
198 "analysis": jsonable(dict(analysis)),
199 "result": {"file": table_file, **jsonable(dict(counts))},
200 "excludes": dict(EXCLUDES),
201 }
202 if exported_at is None:
203 del recipe["exported_at"]
204 return recipe
207def recipe_file_name(table_file: str) -> str:
208 """``cohort_profile_tfd_3.csv`` → ``cohort_profile_tfd_3.recipe.json``."""
209 stem = table_file[:-4] if table_file.lower().endswith(".csv") else table_file
210 return f"{stem}.recipe.json"