Coverage for scanpath_studio/dataset_table.py: 97%

102 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""UX-174: the row model behind 📂 Available datasets. 

2 

3Pure — no Streamlit, no pandas — so what a row *says* can be tested without 

4booting the app. `app.render_dataset_table` builds one :class:`DatasetRow` per 

5dataset and draws it; everything that decides a cell's text lives here: 

6 

7- **Counts or a reason.** A count is shown as a grouped integer (``2,400,788``) 

8 and only when something was actually counted or published. A missing count is 

9 never a ``0``, a ``None`` or a ``NaN`` on screen: it names why it is missing 

10 (:data:`NOT_LOADED`, :data:`NOT_REPORTED`, :data:`NOT_APPLICABLE`), or says 

11 :data:`UNKNOWN` when the evidence does not say which. 

12- **Sorting on the numbers, not the text.** :func:`sort_rows` orders by the 

13 integer value, and a missing value sorts last in either direction, so 

14 formatting a cell can never turn a numeric sort into a lexical one. 

15- **A status of its own.** The **Status** column says whether the dataset can 

16 be opened right now, and how fast — *Loaded*, *Available*, *Needs download* 

17 or *Needs setup* 

18 (:attr:`DatasetRow.status_label`) — and every row says it the same way, 

19 whichever dataset is open (BUG-113). Where a row's numbers came from 

20 (DATA-36's *loaded* vs *published*) is a different question, which the 

21 count columns' header explains (:data:`COUNTS_EXPLANATION`). 

22""" 

23 

24from __future__ import annotations 

25 

26from collections.abc import Iterable, Mapping, Sequence 

27from dataclasses import dataclass, field 

28 

29#: The dataset table's count fields, in display order — and the only field names 

30#: a catalogue entry may publish figures under (DATA-36). A typo would otherwise 

31#: be dropped in silence; `tests/test_dataset_published_counts.py` checks every 

32#: declaration against this tuple. 

33DATASET_COUNT_FIELDS: tuple[str, ...] = ( 

34 "Participants", 

35 "Texts", 

36 "Trials", 

37 "Screens", 

38 "Words", 

39 "Fixations", 

40 "Gaze points", 

41) 

42 

43#: The four counts the table shows. The other three — Screens, Words and Gaze 

44#: points — are in the open dataset's 📊 Stats, under the table. 

45TABLE_COUNT_FIELDS: tuple[str, ...] = ("Participants", "Texts", "Trials", "Fixations") 

46 

47#: The one count kept on a phone-width screen, beside the name and the actions. 

48KEY_COUNT_FIELD = "Participants" 

49 

50NOT_LOADED = "Not counted" 

51NOT_REPORTED = "Not reported" 

52NOT_APPLICABLE = "Not applicable" 

53UNKNOWN = "Unknown" 

54 

55#: Each missing-value label's meaning, for its hover text. 

56GAP_EXPLANATIONS: Mapping[str, str] = { 

57 NOT_LOADED: "Not counted yet — this dataset has not been opened, and it " 

58 "publishes no figures of its own.", 

59 NOT_REPORTED: "The corpus' published figures do not include this one. Open " 

60 "the dataset to count it.", 

61 NOT_APPLICABLE: "This dataset has nothing of this kind to count.", 

62 UNKNOWN: "Not available — the data loaded, but this count could not be " 

63 "determined from it.", 

64} 

65 

66#: Why a loaded dataset can hold no value for a field: the table the count is 

67#: taken from is absent, which is a property of the dataset, not a gap in the 

68#: counting. Every other field left blank by a load is :data:`UNKNOWN`. 

69_NOT_APPLICABLE_WHEN_LOADED: Mapping[str, str] = { 

70 "Screens": "Every trial is a single screen.", 

71 "Words": "This dataset has no Words table.", 

72 "Fixations": "This dataset has no fixation table.", 

73 "Gaze points": "This dataset has no raw-gaze samples.", 

74} 

75 

76#: BUG-113 — the **Status** column's three values. It used to say *Loaded* / 

77#: *Not loaded*, which is where the counts came from, and computed *Needs setup* 

78#: for the open dataset only — so a corpus whose files had gone read *Loaded* 

79#: until you opened it, and *Needs setup* the moment you did. Every row now says 

80#: whether its dataset can be opened. 

81#: Its data being here splits in two: already read this session (opens at 

82#: once) or still to be read (opening reads its files, which takes a while on 

83#: a large corpus). 

84LOADED = "Loaded" 

85AVAILABLE = "Available" 

86NEEDS_DOWNLOAD = "Needs download" 

87NEEDS_SETUP = "Needs setup" 

88 

89#: What each value of the **Status** column means, for its hover text. 

90STATUS_EXPLANATIONS: Mapping[str, str] = { 

91 LOADED: "Read this session, so it usually opens quickly.", 

92 AVAILABLE: "Its files are here; opening it reads them, which can take a " 

93 "while for a large dataset.", 

94 NEEDS_DOWNLOAD: "Its files are not in its folder yet: open it, then click " 

95 "**Download**.", 

96 NEEDS_SETUP: "Its files were not found and there is no download: open it " 

97 "and set its **Data directory** to their folder.", 

98} 

99 

100#: Where a row's numbers come from (DATA-36), for the count columns' header. 

101COUNTS_EXPLANATION = ( 

102 "Counted from a dataset's own rows once it has been opened; until then, the " 

103 "figures the corpus publishes for itself." 

104) 

105 

106#: Kind is ordered by what a row is, not alphabetically, when it is sorted. 

107KIND_ORDER: tuple[str, ...] = ("Demo", "Manual", "Private", "Public") 

108 

109 

110def format_count(value: int) -> str: 

111 """A count with thousands separators — ``2400788`` → ``"2,400,788"``.""" 

112 return f"{int(value):,}" 

113 

114 

115@dataclass(frozen=True) 

116class DatasetRow: 

117 """What one dataset's row of the table shows. 

118 

119 ``source`` is DATA-36's ``"loaded"`` / ``"published"`` / ``""``, and 

120 ``counts`` holds only that source's numbers (a row never mixes the two). 

121 ``measured`` says whether the session ever counted this dataset at all, 

122 which is what tells *Not loaded* from *Unknown* on a row with no numbers. 

123 ``status`` is whether the dataset can be opened now — blank when it can 

124 (then ``loaded`` says :data:`LOADED` or :data:`AVAILABLE`), else 

125 :data:`NEEDS_DOWNLOAD` / :data:`NEEDS_SETUP` — deliberately a 

126 field of its own: it says what the app can do with the dataset right now, 

127 which is a different question from where its numbers came from (BUG-113). 

128 ``order`` is the row's place in the unsorted list, so the list returns to 

129 it and does not move when the open dataset changes. 

130 """ 

131 

132 token: str 

133 name: str 

134 kind: str = "" 

135 language: str = "" 

136 source: str = "" 

137 counts: Mapping[str, int | None] = field(default_factory=dict) 

138 exceeds_published: tuple[str, ...] = () 

139 active: bool = False 

140 measured: bool = False 

141 status: str = "" 

142 loaded: bool = False 

143 order: int = 0 

144 

145 def value(self, count_field: str) -> int | None: 

146 value = self.counts.get(count_field) 

147 return None if value is None else int(value) 

148 

149 def gap(self, count_field: str) -> str: 

150 """Why ``count_field`` has no value — ``""`` when it has one.""" 

151 if self.value(count_field) is not None: 

152 return "" 

153 if self.source == "loaded": 

154 if count_field in _NOT_APPLICABLE_WHEN_LOADED: 

155 return NOT_APPLICABLE 

156 return UNKNOWN 

157 if self.source == "published": 

158 return NOT_REPORTED 

159 return UNKNOWN if self.measured else NOT_LOADED 

160 

161 def gap_explanation(self, count_field: str) -> str: 

162 """The sentence behind :meth:`gap`, specific to the field where it can be.""" 

163 gap = self.gap(count_field) 

164 if gap == NOT_APPLICABLE: 

165 return _NOT_APPLICABLE_WHEN_LOADED[count_field] 

166 return GAP_EXPLANATIONS.get(gap, "") 

167 

168 def cell(self, count_field: str) -> str: 

169 """The cell's text: the grouped count, or the reason there is none.""" 

170 value = self.value(count_field) 

171 return self.gap(count_field) if value is None else format_count(value) 

172 

173 @property 

174 def status_label(self) -> str: 

175 """The **Status** cell — :data:`LOADED` / :data:`AVAILABLE` unless a 

176 missing-files state says otherwise. 

177 

178 Never derived from :attr:`source` (BUG-113): whether the numbers were 

179 counted or published says nothing about whether the files are here — 

180 and counts are remembered across sessions, so neither do they say 

181 whether this session has read the dataset (``loaded``). 

182 """ 

183 return self.status or (LOADED if self.loaded else AVAILABLE) 

184 

185 

186def _text_key(row: DatasetRow, column: str) -> str | int | None: 

187 if column == "Dataset": 

188 return row.name.casefold() or None 

189 if column == "Language": 

190 return row.language.casefold() or None 

191 if column == "Status": 

192 return row.status_label.casefold() 

193 if column == "Kind": 

194 return KIND_ORDER.index(row.kind) if row.kind in KIND_ORDER else None 

195 raise ValueError(f"not a sortable column: {column!r}") 

196 

197 

198def sort_key(row: DatasetRow, column: str): 

199 """The value ``column`` sorts ``row`` on, or ``None`` for a missing one.""" 

200 if column in DATASET_COUNT_FIELDS: 

201 return row.value(column) 

202 return _text_key(row, column) 

203 

204 

205def default_descending(column: str) -> bool: 

206 """A count column sorts largest first; a text column A → Z.""" 

207 return column in DATASET_COUNT_FIELDS 

208 

209 

210def sort_rows( 

211 rows: Iterable[DatasetRow], column: str | None, *, descending: bool = False 

212) -> list[DatasetRow]: 

213 """``rows`` sorted on ``column``'s value, missing values last either way. 

214 

215 ``column=None`` is the unsorted list — the order the datasets are offered 

216 in, which nothing in the table (least of all opening a dataset) changes. 

217 """ 

218 rows = sorted(rows, key=lambda row: row.order) 

219 if column is None: 

220 return rows 

221 present = [row for row in rows if sort_key(row, column) is not None] 

222 missing = [row for row in rows if sort_key(row, column) is None] 

223 present.sort(key=lambda row: sort_key(row, column), reverse=descending) 

224 return present + missing 

225 

226 

227def next_sort(current: tuple[str, bool] | None, column: str) -> tuple[str, bool] | None: 

228 """A header click: default direction → reversed → back to unsorted.""" 

229 first = default_descending(column) 

230 if current is None or current[0] != column: 

231 return (column, first) 

232 if current[1] == first: 

233 return (column, not first) 

234 return None 

235 

236 

237def filter_rows( 

238 rows: Iterable[DatasetRow], 

239 *, 

240 query: str = "", 

241 kinds: Sequence[str] = (), 

242 languages: Sequence[str] = (), 

243) -> list[DatasetRow]: 

244 """The rows matching a name search and the Kind / Language picks. 

245 

246 An empty pick means "any", so the filters narrow only once something is 

247 chosen. The search is a case-insensitive substring of the name. 

248 """ 

249 needle = query.strip().casefold() 

250 return [ 

251 row 

252 for row in rows 

253 if (not needle or needle in row.name.casefold()) 

254 and (not kinds or row.kind in kinds) 

255 and (not languages or row.language in languages) 

256 ] 

257 

258 

259def row_record(row: DatasetRow) -> dict: 

260 """``row`` as a flat record — the table's inspection seam for tests. 

261 

262 Keys follow the columns' names; a count is the integer or ``None`` (the 

263 *value*, never its formatted cell), and ``Counts`` is DATA-36's badge word. 

264 """ 

265 record = { 

266 "Kind": row.kind, 

267 "Dataset": row.name, 

268 "Language": row.language, 

269 "Counts": {"loaded": "Loaded", "published": "Published"}.get(row.source, ""), 

270 } 

271 for count_field in DATASET_COUNT_FIELDS: 

272 record[count_field] = row.value(count_field) 

273 record["Status"] = row.status_label 

274 record["_token"] = row.token 

275 record["_active"] = row.active 

276 record["_cells"] = {f: row.cell(f) for f in DATASET_COUNT_FIELDS} 

277 return record