Coverage for scanpath_studio/synthetic.py: 100%
63 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""A small, fully-specified synthetic scanpath — ground truth for testing + a
2visualizable in-app data source ("Synthetic test trial").
4Real eye-tracking exports are noisy and their reading measures can only be
5checked approximately. This module ships a tiny hand-built trial whose geometry
6and reading sequence are simple enough that *every* canonical measure has an
7exact, hand-traced expected value (see ``EXPECTED``). The app exposes it as a
8data source so you can eyeball the scanpath against that ground truth; the test
9suite (``tests/test_synthetic.py``) asserts the measures match.
11Layout — 6 words, 2 lines of 3, 50x20 px boxes on a clean grid::
13 x=100..150 x=200..250 x=300..350
14 y=100..120 [0 The] [1 cat] [2 sat] (line 0, y-center 110)
15 y=200..220 [3 on ] [4 the] [5 mat] (line 1, y-center 210)
17Reading sequence (9 fixations) — first-pass refixation on word 0, a regression
18from word 2 back to word 1, and one deliberately out-of-text fixation::
20 # t(ms) (x, y) dur lands on
21 1 0 (115,110) 100 word 0 (first fixation)
22 2 100 (135,110) 50 word 0 (refixation, same first pass)
23 3 150 (225,110) 150 word 1
24 4 300 (325,110) 120 word 2
25 5 420 (225,110) 80 word 1 (regression back)
26 6 500 (125,210) 200 word 3
27 7 700 (225,210) 90 word 4
28 8 790 (700,700) 60 -- (out of text; word_id NaN)
29 9 850 (325,210) 110 word 5
30"""
32from __future__ import annotations
34import io
35import zipfile
37import numpy as np
38import pandas as pd
40# Human-friendly ids so the trial reads clearly in the app's trial picker.
41PARTICIPANT = "synthetic"
42TRIAL = "synthetic_2line_demo"
43PARAGRAPH = "synthetic_2line_demo"
45_WORD_TEXT = ["The", "cat", "sat", "on", "the", "mat"]
46_WORD_X = [100, 200, 300, 100, 200, 300]
47_WORD_Y = [100, 100, 100, 200, 200, 200]
48_BOX_W = 50
49_BOX_H = 20
52def make_synthetic_words() -> pd.DataFrame:
53 """Return the normalized words/IA frame for the synthetic trial."""
54 n = len(_WORD_TEXT)
55 return pd.DataFrame(
56 {
57 "participant_id": [PARTICIPANT] * n,
58 "trial_id": [TRIAL] * n,
59 "text_id": [PARAGRAPH] * n,
60 "word_id": list(range(n)),
61 "text": list(_WORD_TEXT),
62 # line_idx is intentionally a constant: real OneStop IA exports do
63 # not carry a usable per-word line number, so the by-line feature
64 # must infer lines from geometry, not trust this column.
65 "line_idx": [1] * n,
66 "x": list(_WORD_X),
67 "y": list(_WORD_Y),
68 "width": [_BOX_W] * n,
69 "height": [_BOX_H] * n,
70 }
71 )
74# (x, y, duration_ms, timestamp_ms) for each fixation in temporal order.
75_FIX_ROWS = [
76 (115, 110, 100, 0),
77 (135, 110, 50, 100),
78 (225, 110, 150, 150),
79 (325, 110, 120, 300),
80 (225, 110, 80, 420),
81 (125, 210, 200, 500),
82 (225, 210, 90, 700),
83 (700, 700, 60, 790),
84 (325, 210, 110, 850),
85]
88def make_synthetic_fixations() -> pd.DataFrame:
89 """Return the normalized fixations frame for the synthetic trial.
91 ``word_id`` is left NaN on purpose so that fixation-to-word assignment runs
92 and can itself be tested / exercised.
93 """
94 n = len(_FIX_ROWS)
95 return pd.DataFrame(
96 {
97 "participant_id": [PARTICIPANT] * n,
98 "trial_id": [TRIAL] * n,
99 "text_id": [PARAGRAPH] * n,
100 "x": [r[0] for r in _FIX_ROWS],
101 "y": [r[1] for r in _FIX_ROWS],
102 "duration_ms": [r[2] for r in _FIX_ROWS],
103 "timestamp_ms": [r[3] for r in _FIX_ROWS],
104 "word_id": [np.nan] * n,
105 "order_in_trial": list(range(1, n + 1)),
106 "pass_index": [1] * n,
107 }
108 )
111def load_synthetic_data() -> tuple[pd.DataFrame, pd.DataFrame]:
112 """(words, fixations) for the "Synthetic test trial" data source."""
113 return make_synthetic_words(), make_synthetic_fixations()
116# Hand-traced expected values. Keyed by word_id (0..5) unless noted.
117EXPECTED = {
118 # Word-box line clustering (top-to-bottom, 0-based), per word_id 0..5.
119 "word_line": [0, 0, 0, 1, 1, 1],
120 # Fixation -> word_id via bbox containment (NaN for the out-of-text fix).
121 "fixation_word_id": [0, 0, 1, 2, 1, 3, 4, np.nan, 5],
122 # Fixation -> nearest text line (the out-of-text fix snaps to line 1).
123 "fixation_line": [0, 0, 0, 0, 0, 1, 1, 1, 1],
124 # In-text mask (True = inside some word box): exactly one out-of-text fix.
125 "in_text": [True, True, True, True, True, True, True, False, True],
126 "out_of_text_count": 1,
127 # Per-word reading measures.
128 "first_fixation_ms": {0: 100, 1: 150, 2: 120, 3: 200, 4: 90, 5: 110},
129 "first_pass_gaze_duration_ms": {0: 150, 1: 150, 2: 120, 3: 200, 4: 90, 5: 110},
130 "regression_path_duration_ms": {0: 150, 1: 150, 2: 200, 3: 200, 4: 90, 5: 110},
131 "total_fixation_duration_ms": {0: 150, 1: 230, 2: 120, 3: 200, 4: 90, 5: 110},
132 "n_fixations": {0: 2, 1: 2, 2: 1, 3: 1, 4: 1, 5: 1},
133 "skip_flag": {0: False, 1: False, 2: False, 3: False, 4: False, 5: False},
134 "regression_in_flag": {0: False, 1: True, 2: False, 3: False, 4: False, 5: False},
135 "regression_out_flag": {0: False, 1: False, 2: True, 3: False, 4: False, 5: False},
136}
139# --- DATA-67: the example import pair ---------------------------------------
140#
141# The add-dataset wizard's **Download example tables** hands a new user the
142# trial above as the two files an import starts from, with every column named
143# the way the wizard's auto-detection reads it, so the pair maps without a
144# single manual pick (`tests/test_example_tables.py` holds it to that). The
145# column notes are the README's, kept here beside the frames they describe.
147#: The AOI table's columns, in file order, with what each one holds.
148EXAMPLE_AOI_COLUMNS: dict[str, str] = {
149 "participant_id": "who read the text.",
150 "trial_id": "one trial (one reading of a text); participant_id + trial_id together "
151 "identify it, in both tables.",
152 "text_id": "which text was shown; shared by everyone who read it.",
153 "word_id": "the word's number within its text, from 0.",
154 "text": "the word as displayed.",
155 "x": "left edge of the word's box, in screen pixels.",
156 "y": "top edge of the word's box, in screen pixels.",
157 "width": "box width, in pixels.",
158 "height": "box height, in pixels.",
159 "first_fixation_ms": "optional reading measure: first fixation duration, ms.",
160 "total_fixation_duration_ms": "optional reading measure: total fixation "
161 "duration (dwell time), ms.",
162}
164#: The fixation table's columns, in file order, with what each one holds.
165EXAMPLE_FIXATION_COLUMNS: dict[str, str] = {
166 "participant_id": "who read the text; matches the Words table.",
167 "trial_id": "which trial; matches the Words table.",
168 "text_id": "which text was shown; matches the Words table.",
169 "timestamp_ms": "fixation onset, in ms from the start of the trial.",
170 "duration_ms": "fixation duration, in ms.",
171 "x": "horizontal gaze position, in screen pixels.",
172 "y": "vertical gaze position, in screen pixels.",
173}
175EXAMPLE_AOI_FILE = "aois.csv"
176EXAMPLE_FIXATION_FILE = "fixations.csv"
177EXAMPLE_ZIP_FILE = "scanpath_studio_example_tables.zip"
180def example_import_tables() -> tuple[pd.DataFrame, pd.DataFrame]:
181 """The synthetic trial as an AOI table and a fixation table to import.
183 The AOI table carries two of the trial's hand-traced reading measures
184 (``EXPECTED``) so the example shows that a dataset brings its own: the app
185 computes none for Corpus Analysis (AN-32). The fixation table has no word
186 ids: the app assigns each fixation to the word box it falls in.
187 """
188 words = make_synthetic_words()
189 for column in ("first_fixation_ms", "total_fixation_duration_ms"):
190 words[column] = words["word_id"].map(EXPECTED[column])
191 fixations = make_synthetic_fixations()
192 return (
193 words[list(EXAMPLE_AOI_COLUMNS)],
194 fixations[list(EXAMPLE_FIXATION_COLUMNS)],
195 )
198def _column_notes(columns: dict[str, str]) -> str:
199 return "\n".join(f"- `{name}`: {note}" for name, note in columns.items())
202def example_readme() -> str:
203 """The example's README: what the two files are, their units and IDs."""
204 words, fixations = example_import_tables()
205 return f"""# Scanpath Studio example tables
207One participant reading a six-word text on two lines ("The cat sat / on the mat")
208once. On **Add dataset**, upload `{EXAMPLE_AOI_FILE}` as **Words** and
209`{EXAMPLE_FIXATION_FILE}` as **Fixations**; every column maps automatically.
210Unzip first: a zip dropped on one upload box is read as one table.
212Units: positions are screen pixels from the top-left corner, with y growing
213downward. Times are milliseconds.
215## {EXAMPLE_AOI_FILE}: one row per word ({len(words)} rows)
217{_column_notes(EXAMPLE_AOI_COLUMNS)}
219## {EXAMPLE_FIXATION_FILE}: one row per fixation, in time order ({len(fixations)} rows)
221{_column_notes(EXAMPLE_FIXATION_COLUMNS)}
223The fixation table has no word column: each fixation is assigned to the word
224box it falls in. The fixation at (700, 700) is outside every box on purpose, and counts as out of text.
225"""
228def example_import_zip() -> bytes:
229 """``aois.csv`` + ``fixations.csv`` + ``README.md``, zipped, for download."""
230 words, fixations = example_import_tables()
231 buffer = io.BytesIO()
232 with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive:
233 archive.writestr(EXAMPLE_AOI_FILE, words.to_csv(index=False))
234 archive.writestr(EXAMPLE_FIXATION_FILE, fixations.to_csv(index=False))
235 archive.writestr("README.md", example_readme())
236 return buffer.getvalue()
239def make_multipart_synthetic_data() -> tuple[pd.DataFrame, pd.DataFrame]:
240 """Two-screen executable specification for multipart trials (DATA-21).
242 Screen 1 contains a regression (word 2 → word 1); screen 2 has a different
243 canvas and resets the screen-local fixation index while the parent clock and
244 global fixation id remain monotonic. The fixture is intentionally tiny so
245 grouping, navigation, animation boundaries, and export paths have exact
246 hand-authored expectations.
247 """
248 words = pd.DataFrame(
249 [
250 ("intro", 1, 0, "Read", 50, 50, 70, 20, 640, 480),
251 ("intro", 1, 1, "this", 140, 50, 60, 20, 640, 480),
252 ("intro", 1, 2, "first", 220, 50, 70, 20, 640, 480),
253 ("question", 2, 0, "Answer", 100, 120, 90, 24, 800, 600),
254 ("question", 2, 1, "now", 215, 120, 55, 24, 800, 600),
255 ],
256 columns=[
257 "screen_id",
258 "screen_index",
259 "word_id",
260 "text",
261 "x",
262 "y",
263 "width",
264 "height",
265 "canvas_width",
266 "canvas_height",
267 ],
268 )
269 words.insert(0, "text_id", ["intro"] * 3 + ["question"] * 2)
270 words.insert(0, "trial_id", "multipart_demo")
271 words.insert(0, "participant_id", "synthetic")
272 words["line_idx"] = 1
274 fixations = pd.DataFrame(
275 [
276 ("intro", 1, 1, 1, 0, 0, 75, 60, 100, 0),
277 ("intro", 1, 2, 2, 100, 100, 165, 60, 110, 1),
278 ("intro", 1, 3, 3, 210, 210, 245, 60, 90, 2),
279 ("intro", 1, 4, 4, 300, 300, 165, 60, 80, 1),
280 ("question", 2, 5, 1, 1_000, 0, 145, 132, 120, 0),
281 ("question", 2, 6, 2, 1_120, 120, 240, 132, 100, 1),
282 ],
283 columns=[
284 "screen_id",
285 "screen_index",
286 "fixation_id",
287 "screen_fixation_id",
288 "timestamp_ms",
289 "screen_timestamp_ms",
290 "x",
291 "y",
292 "duration_ms",
293 "word_id",
294 ],
295 )
296 fixations.insert(0, "text_id", ["intro"] * 4 + ["question"] * 2)
297 fixations.insert(0, "trial_id", "multipart_demo")
298 fixations.insert(0, "participant_id", "synthetic")
299 fixations["order_in_trial"] = range(1, len(fixations) + 1)
300 fixations["order_in_screen"] = [1, 2, 3, 4, 1, 2]
301 fixations["canvas_width"] = [640] * 4 + [800] * 2
302 fixations["canvas_height"] = [480] * 4 + [600] * 2
303 return words, fixations
306MULTIPART_EXPECTED = {
307 "screens": ["intro", "question"],
308 "screen_indexes": [1, 2],
309 "canvas_sizes": [(640, 480), (800, 600)],
310 "fixations_per_screen": [4, 2],
311 "intro_regression_destination": 1,
312 "parent_timestamp_span": (0, 1_120),
313}