Coverage for scanpath_studio/analysis_recipe.py: 88%

78 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""AN-34: the analysis recipe — how one Corpus Analysis table was made. 

2 

3A small JSON file downloaded beside each Corpus Analysis table. It records what 

4the table cannot say about itself: the app version, the dataset it was read 

5from, the trial filters that shaped the pool, the analysis choices (text, 

6screen, measure, aggregation, normalization, spread, minimum readers, group 

7definitions) and the counts behind the result. 

8 

9It **references** the dataset by name and never embeds it: no table rows, and 

10no annotation notes (a *Favorites only* or tag filter is recorded as the filter 

11it is, not as the annotations behind it). Filters and figure settings are kept 

12apart on purpose — this file holds the filters and no styling; the figure 

13settings file (🔗 Share → File) holds styling and no filters — so the recipe 

14says so in ``excludes`` rather than leaving a reader to guess. 

15 

16It describes the current analysis only. There is no runner and no workspace 

17format: nothing reads a recipe back. 

18 

19Pure — no Streamlit; ``tabs._download_tidy`` assembles the inputs. 

20""" 

21 

22from __future__ import annotations 

23 

24from collections.abc import Mapping, Sequence 

25from typing import Any 

26 

27import numpy as np 

28import pandas as pd 

29 

30#: What a recipe is, so a script can tell it from the figure settings file. 

31RECIPE_KIND = "scanpath-studio/analysis-recipe" 

32#: Bumped when a field changes meaning; adding a field does not bump it. 

33RECIPE_VERSION = 1 

34 

35#: What the recipe deliberately leaves out, written into every one. 

36EXCLUDES = { 

37 "figure_settings": ( 

38 "Palette, canvas and fonts live in the settings file (Scanpath → " 

39 "Share → File), which holds no trial filters." 

40 ), 

41 "data": ( 

42 "No table rows and no annotation notes. The dataset is referenced by " 

43 "name and must be loaded to repeat the analysis." 

44 ), 

45} 

46 

47 

48def jsonable(value: Any) -> Any: 

49 """``value`` with sets, tuples and numpy scalars made JSON-native. 

50 

51 Sets are sorted (by their text) so the same analysis writes the same file. 

52 """ 

53 if isinstance(value, Mapping): 

54 return {str(k): jsonable(v) for k, v in value.items()} 

55 if isinstance(value, (set, frozenset)): 

56 return [jsonable(v) for v in sorted(value, key=str)] 

57 if isinstance(value, (list, tuple)): 

58 return [jsonable(v) for v in value] 

59 if isinstance(value, (np.ndarray, pd.Index, pd.Series)): 

60 return [jsonable(v) for v in value.tolist()] 

61 if isinstance(value, np.bool_): 

62 return bool(value) 

63 if isinstance(value, np.integer): 

64 return int(value) 

65 if isinstance(value, np.floating): 

66 return None if np.isnan(value) else float(value) 

67 if isinstance(value, float) and np.isnan(value): 

68 return None 

69 if value is None or isinstance(value, (str, int, float, bool)): 

70 return value 

71 return str(value) 

72 

73 

74def group_definition(label: str, spec: Mapping | None) -> dict: 

75 """One cohort as ``{"label", "constraints"}`` — the spec the views used. 

76 

77 ``spec`` is ``aggregation.group_mask``'s ``{column: allowed values}``; a 

78 composite key (a trial-metadata cohort's ``(participant_id, trial_id)``) 

79 keeps its columns as a list. An empty spec is the whole pool. 

80 """ 

81 constraints = [ 

82 { 

83 "field": list(col) if isinstance(col, tuple) else str(col), 

84 "values": jsonable(values), 

85 } 

86 for col, values in (spec or {}).items() 

87 if values is not None 

88 ] 

89 return {"label": str(label), "constraints": constraints} 

90 

91 

92def analysis_choices( 

93 *, 

94 section: str | None = None, 

95 view: str | None = None, 

96 text: tuple[str, Any] | None = None, 

97 screen: Any = None, 

98 reader: Any = None, 

99 measure: Any = None, 

100 measures: Sequence[Any] | None = None, 

101 aggregation: str | None = None, 

102 normalize: bool | None = None, 

103 spread: str | None = None, 

104 min_readers: int | None = None, 

105 feature: str | None = None, 

106 x_axis: str | None = None, 

107 groups: Sequence[dict] | None = None, 

108) -> dict: 

109 """The choices that made one table, with what does not apply left out. 

110 

111 ``measure`` is an ``aggregation.Measure`` (anything with ``key`` / ``label`` 

112 / ``is_rate``). ``normalize`` is recorded as it was *applied*: the 

113 aggregation helpers never z-score a 0–1 rate, so a rate reads ``none`` 

114 whatever the toggle says. 

115 """ 

116 out: dict = {} 

117 if section: 

118 out["section"] = section 

119 if view: 

120 out["view"] = view 

121 if text is not None: 

122 out["text"] = {"field": str(text[0]), "id": jsonable(text[1])} 

123 if screen is not None: 

124 out["screen"] = jsonable(screen) 

125 if reader is not None: 

126 out["reader"] = jsonable(reader) 

127 if measure is not None: 

128 out["measure"] = {"key": measure.key, "label": measure.label} 

129 if measures: 

130 out["measures"] = [{"key": m.key, "label": m.label} for m in measures] 

131 if aggregation: 

132 out["aggregation"] = aggregation 

133 if normalize is not None: 

134 rate = bool(getattr(measure, "is_rate", False)) 

135 out["normalization"] = ( 

136 "z-score per participant" if normalize and not rate else "none" 

137 ) 

138 if spread: 

139 out["spread"] = spread 

140 if min_readers is not None: 

141 out["min_readers"] = int(min_readers) 

142 if feature: 

143 out["feature"] = feature 

144 if x_axis: 

145 out["x_axis"] = x_axis 

146 if groups: 

147 out["groups"] = list(groups) 

148 return out 

149 

150 

151def result_counts(table: pd.DataFrame | None, extra: Mapping | None = None) -> dict: 

152 """What the table itself says about its size: rows, and readers when it 

153 names them. ``extra`` adds counts the view already computed (a cohort's 

154 readers and fixations), never recomputed here.""" 

155 out: dict = {"rows": 0 if table is None else len(table)} 

156 if table is not None and "participant_id" in table.columns: 

157 out["readers"] = int(table["participant_id"].astype(str).nunique()) 

158 if extra: 

159 out.update(jsonable(dict(extra))) 

160 return out 

161 

162 

163def build_analysis_recipe( 

164 *, 

165 app_version: str, 

166 dataset: Mapping, 

167 trial_filters: Sequence[Mapping], 

168 pool: Mapping, 

169 analysis: Mapping, 

170 table_file: str, 

171 counts: Mapping, 

172 exported_at: str | None = None, 

173) -> dict: 

174 """The recipe for one Corpus Analysis table. 

175 

176 ``trial_filters`` is ``controls.active_filter_items`` — one 

177 ``{"field", "values" | "range"}`` entry per filter narrowing the pool (a 

178 range also carries ``"unknown": "kept" | "excluded"`` — what it does with 

179 the records that have no value); an 

180 empty list is an unfiltered pool. ``pool`` holds its trial and reader 

181 counts against the dataset's. 

182 """ 

183 recipe = { 

184 "kind": RECIPE_KIND, 

185 "version": RECIPE_VERSION, 

186 "app": {"name": "Scanpath Studio", "version": str(app_version)}, 

187 "exported_at": exported_at, 

188 "dataset": jsonable(dict(dataset)), 

189 "trial_filters": [ 

190 { 

191 k: jsonable(v) 

192 for k, v in item.items() 

193 if k in ("field", "values", "range", "unknown") 

194 } 

195 for item in trial_filters 

196 ], 

197 "pool": jsonable(dict(pool)), 

198 "analysis": jsonable(dict(analysis)), 

199 "result": {"file": table_file, **jsonable(dict(counts))}, 

200 "excludes": dict(EXCLUDES), 

201 } 

202 if exported_at is None: 

203 del recipe["exported_at"] 

204 return recipe 

205 

206 

207def recipe_file_name(table_file: str) -> str: 

208 """``cohort_profile_tfd_3.csv`` → ``cohort_profile_tfd_3.recipe.json``.""" 

209 stem = table_file[:-4] if table_file.lower().endswith(".csv") else table_file 

210 return f"{stem}.recipe.json"