Coverage for scanpath_studio/dataset_table.py: 97%
102 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""UX-174: the row model behind 📂 Available datasets.
3Pure — no Streamlit, no pandas — so what a row *says* can be tested without
4booting the app. `app.render_dataset_table` builds one :class:`DatasetRow` per
5dataset and draws it; everything that decides a cell's text lives here:
7- **Counts or a reason.** A count is shown as a grouped integer (``2,400,788``)
8 and only when something was actually counted or published. A missing count is
9 never a ``0``, a ``None`` or a ``NaN`` on screen: it names why it is missing
10 (:data:`NOT_LOADED`, :data:`NOT_REPORTED`, :data:`NOT_APPLICABLE`), or says
11 :data:`UNKNOWN` when the evidence does not say which.
12- **Sorting on the numbers, not the text.** :func:`sort_rows` orders by the
13 integer value, and a missing value sorts last in either direction, so
14 formatting a cell can never turn a numeric sort into a lexical one.
15- **A status of its own.** The **Status** column says whether the dataset can
16 be opened right now, and how fast — *Loaded*, *Available*, *Needs download*
17 or *Needs setup*
18 (:attr:`DatasetRow.status_label`) — and every row says it the same way,
19 whichever dataset is open (BUG-113). Where a row's numbers came from
20 (DATA-36's *loaded* vs *published*) is a different question, which the
21 count columns' header explains (:data:`COUNTS_EXPLANATION`).
22"""
24from __future__ import annotations
26from collections.abc import Iterable, Mapping, Sequence
27from dataclasses import dataclass, field
29#: The dataset table's count fields, in display order — and the only field names
30#: a catalogue entry may publish figures under (DATA-36). A typo would otherwise
31#: be dropped in silence; `tests/test_dataset_published_counts.py` checks every
32#: declaration against this tuple.
33DATASET_COUNT_FIELDS: tuple[str, ...] = (
34 "Participants",
35 "Texts",
36 "Trials",
37 "Screens",
38 "Words",
39 "Fixations",
40 "Gaze points",
41)
43#: The four counts the table shows. The other three — Screens, Words and Gaze
44#: points — are in the open dataset's 📊 Stats, under the table.
45TABLE_COUNT_FIELDS: tuple[str, ...] = ("Participants", "Texts", "Trials", "Fixations")
47#: The one count kept on a phone-width screen, beside the name and the actions.
48KEY_COUNT_FIELD = "Participants"
50NOT_LOADED = "Not counted"
51NOT_REPORTED = "Not reported"
52NOT_APPLICABLE = "Not applicable"
53UNKNOWN = "Unknown"
55#: Each missing-value label's meaning, for its hover text.
56GAP_EXPLANATIONS: Mapping[str, str] = {
57 NOT_LOADED: "Not counted yet — this dataset has not been opened, and it "
58 "publishes no figures of its own.",
59 NOT_REPORTED: "The corpus' published figures do not include this one. Open "
60 "the dataset to count it.",
61 NOT_APPLICABLE: "This dataset has nothing of this kind to count.",
62 UNKNOWN: "Not available — the data loaded, but this count could not be "
63 "determined from it.",
64}
66#: Why a loaded dataset can hold no value for a field: the table the count is
67#: taken from is absent, which is a property of the dataset, not a gap in the
68#: counting. Every other field left blank by a load is :data:`UNKNOWN`.
69_NOT_APPLICABLE_WHEN_LOADED: Mapping[str, str] = {
70 "Screens": "Every trial is a single screen.",
71 "Words": "This dataset has no Words table.",
72 "Fixations": "This dataset has no fixation table.",
73 "Gaze points": "This dataset has no raw-gaze samples.",
74}
76#: BUG-113 — the **Status** column's three values. It used to say *Loaded* /
77#: *Not loaded*, which is where the counts came from, and computed *Needs setup*
78#: for the open dataset only — so a corpus whose files had gone read *Loaded*
79#: until you opened it, and *Needs setup* the moment you did. Every row now says
80#: whether its dataset can be opened.
81#: Its data being here splits in two: already read this session (opens at
82#: once) or still to be read (opening reads its files, which takes a while on
83#: a large corpus).
84LOADED = "Loaded"
85AVAILABLE = "Available"
86NEEDS_DOWNLOAD = "Needs download"
87NEEDS_SETUP = "Needs setup"
89#: What each value of the **Status** column means, for its hover text.
90STATUS_EXPLANATIONS: Mapping[str, str] = {
91 LOADED: "Read this session, so it usually opens quickly.",
92 AVAILABLE: "Its files are here; opening it reads them, which can take a "
93 "while for a large dataset.",
94 NEEDS_DOWNLOAD: "Its files are not in its folder yet: open it, then click "
95 "**Download**.",
96 NEEDS_SETUP: "Its files were not found and there is no download: open it "
97 "and set its **Data directory** to their folder.",
98}
100#: Where a row's numbers come from (DATA-36), for the count columns' header.
101COUNTS_EXPLANATION = (
102 "Counted from a dataset's own rows once it has been opened; until then, the "
103 "figures the corpus publishes for itself."
104)
106#: Kind is ordered by what a row is, not alphabetically, when it is sorted.
107KIND_ORDER: tuple[str, ...] = ("Demo", "Manual", "Private", "Public")
110def format_count(value: int) -> str:
111 """A count with thousands separators — ``2400788`` → ``"2,400,788"``."""
112 return f"{int(value):,}"
115@dataclass(frozen=True)
116class DatasetRow:
117 """What one dataset's row of the table shows.
119 ``source`` is DATA-36's ``"loaded"`` / ``"published"`` / ``""``, and
120 ``counts`` holds only that source's numbers (a row never mixes the two).
121 ``measured`` says whether the session ever counted this dataset at all,
122 which is what tells *Not loaded* from *Unknown* on a row with no numbers.
123 ``status`` is whether the dataset can be opened now — blank when it can
124 (then ``loaded`` says :data:`LOADED` or :data:`AVAILABLE`), else
125 :data:`NEEDS_DOWNLOAD` / :data:`NEEDS_SETUP` — deliberately a
126 field of its own: it says what the app can do with the dataset right now,
127 which is a different question from where its numbers came from (BUG-113).
128 ``order`` is the row's place in the unsorted list, so the list returns to
129 it and does not move when the open dataset changes.
130 """
132 token: str
133 name: str
134 kind: str = ""
135 language: str = ""
136 source: str = ""
137 counts: Mapping[str, int | None] = field(default_factory=dict)
138 exceeds_published: tuple[str, ...] = ()
139 active: bool = False
140 measured: bool = False
141 status: str = ""
142 loaded: bool = False
143 order: int = 0
145 def value(self, count_field: str) -> int | None:
146 value = self.counts.get(count_field)
147 return None if value is None else int(value)
149 def gap(self, count_field: str) -> str:
150 """Why ``count_field`` has no value — ``""`` when it has one."""
151 if self.value(count_field) is not None:
152 return ""
153 if self.source == "loaded":
154 if count_field in _NOT_APPLICABLE_WHEN_LOADED:
155 return NOT_APPLICABLE
156 return UNKNOWN
157 if self.source == "published":
158 return NOT_REPORTED
159 return UNKNOWN if self.measured else NOT_LOADED
161 def gap_explanation(self, count_field: str) -> str:
162 """The sentence behind :meth:`gap`, specific to the field where it can be."""
163 gap = self.gap(count_field)
164 if gap == NOT_APPLICABLE:
165 return _NOT_APPLICABLE_WHEN_LOADED[count_field]
166 return GAP_EXPLANATIONS.get(gap, "")
168 def cell(self, count_field: str) -> str:
169 """The cell's text: the grouped count, or the reason there is none."""
170 value = self.value(count_field)
171 return self.gap(count_field) if value is None else format_count(value)
173 @property
174 def status_label(self) -> str:
175 """The **Status** cell — :data:`LOADED` / :data:`AVAILABLE` unless a
176 missing-files state says otherwise.
178 Never derived from :attr:`source` (BUG-113): whether the numbers were
179 counted or published says nothing about whether the files are here —
180 and counts are remembered across sessions, so neither do they say
181 whether this session has read the dataset (``loaded``).
182 """
183 return self.status or (LOADED if self.loaded else AVAILABLE)
186def _text_key(row: DatasetRow, column: str) -> str | int | None:
187 if column == "Dataset":
188 return row.name.casefold() or None
189 if column == "Language":
190 return row.language.casefold() or None
191 if column == "Status":
192 return row.status_label.casefold()
193 if column == "Kind":
194 return KIND_ORDER.index(row.kind) if row.kind in KIND_ORDER else None
195 raise ValueError(f"not a sortable column: {column!r}")
198def sort_key(row: DatasetRow, column: str):
199 """The value ``column`` sorts ``row`` on, or ``None`` for a missing one."""
200 if column in DATASET_COUNT_FIELDS:
201 return row.value(column)
202 return _text_key(row, column)
205def default_descending(column: str) -> bool:
206 """A count column sorts largest first; a text column A → Z."""
207 return column in DATASET_COUNT_FIELDS
210def sort_rows(
211 rows: Iterable[DatasetRow], column: str | None, *, descending: bool = False
212) -> list[DatasetRow]:
213 """``rows`` sorted on ``column``'s value, missing values last either way.
215 ``column=None`` is the unsorted list — the order the datasets are offered
216 in, which nothing in the table (least of all opening a dataset) changes.
217 """
218 rows = sorted(rows, key=lambda row: row.order)
219 if column is None:
220 return rows
221 present = [row for row in rows if sort_key(row, column) is not None]
222 missing = [row for row in rows if sort_key(row, column) is None]
223 present.sort(key=lambda row: sort_key(row, column), reverse=descending)
224 return present + missing
227def next_sort(current: tuple[str, bool] | None, column: str) -> tuple[str, bool] | None:
228 """A header click: default direction → reversed → back to unsorted."""
229 first = default_descending(column)
230 if current is None or current[0] != column:
231 return (column, first)
232 if current[1] == first:
233 return (column, not first)
234 return None
237def filter_rows(
238 rows: Iterable[DatasetRow],
239 *,
240 query: str = "",
241 kinds: Sequence[str] = (),
242 languages: Sequence[str] = (),
243) -> list[DatasetRow]:
244 """The rows matching a name search and the Kind / Language picks.
246 An empty pick means "any", so the filters narrow only once something is
247 chosen. The search is a case-insensitive substring of the name.
248 """
249 needle = query.strip().casefold()
250 return [
251 row
252 for row in rows
253 if (not needle or needle in row.name.casefold())
254 and (not kinds or row.kind in kinds)
255 and (not languages or row.language in languages)
256 ]
259def row_record(row: DatasetRow) -> dict:
260 """``row`` as a flat record — the table's inspection seam for tests.
262 Keys follow the columns' names; a count is the integer or ``None`` (the
263 *value*, never its formatted cell), and ``Counts`` is DATA-36's badge word.
264 """
265 record = {
266 "Kind": row.kind,
267 "Dataset": row.name,
268 "Language": row.language,
269 "Counts": {"loaded": "Loaded", "published": "Published"}.get(row.source, ""),
270 }
271 for count_field in DATASET_COUNT_FIELDS:
272 record[count_field] = row.value(count_field)
273 record["Status"] = row.status_label
274 record["_token"] = row.token
275 record["_active"] = row.active
276 record["_cells"] = {f: row.cell(f) for f in DATASET_COUNT_FIELDS}
277 return record