Coverage for scanpath_studio/datasets.py: 93%

959 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Loaders for public eye-tracking-while-reading corpora. 

2 

3Currently: PoTeC (Potsdam Textbook Corpus, Jakobi et al. 2024, 

4https://github.com/DiLi-Lab/PoTeC) — a German corpus of 75 readers × 12 

5textbook texts — and MultiplEYE (https://multipleye.eu), a large multilingual 

6corpus whose per-session, filename-keyed shape :func:`load_multipleye` adapts 

7to the generic pipeline. Between them they exercise the two dataset shapes the 

8generic pipeline supports: 

9 

10* **multi-file** — fixations ship as one TSV per reader × text 

11 (``reader0_b0_scanpath.tsv`` … 900 files), concatenated on load; 

12* **stimulus-level AoIs** — word bounding boxes ship once per *text* with no 

13 participant column, and are broadcast across readers 

14 (``data.broadcast_stimulus_words``). 

15 

16PoTeC fixations carry no pixel coordinates — only the fixated character's 

17index. The loader reconstructs (x, y) as the center of that character's 

18bounding box from the per-text ``.ias`` AOI files, giving within-word landing 

19positions. (For AOI-sequence datasets *without* character AOIs, the generic 

20fallback places fixations at word-box centers instead.) 

21 

22Typical use:: 

23 

24 from scanpath_studio.datasets import load_potec 

25 

26 words, fixations = load_potec("data/PoTeC", download=True) 

27 fig = scanpath_studio.plot_scanpath(words, fixations, participant="0", trial="0_b0") 

28""" 

29 

30from __future__ import annotations 

31 

32import contextlib 

33import http.client 

34import io 

35import json 

36import logging 

37import os 

38import re 

39import shutil 

40import ssl 

41import urllib.request 

42import zipfile 

43from collections.abc import Callable, Iterable 

44from dataclasses import dataclass 

45from pathlib import Path 

46 

47import pandas as pd 

48import truststore 

49 

50from . import progress 

51 

52_LOGGER = logging.getLogger(__name__) 

53 

54# PoTeC text p3 contains the German word "null" — pandas' default NA list 

55# would turn it into NaN (see the PoTeC README), so every PoTeC table is read 

56# with keep_default_na=False and this explicit list. 

57_POTEC_NA_VALUES = [ 

58 "#N/A", 

59 "#N/A N/A", 

60 "#NA", 

61 "-1.#IND", 

62 "-1.#QNAN", 

63 "-NaN", 

64 "-nan", 

65 "1.#IND", 

66 "1.#QNAN", 

67 "<NA>", 

68 "N/A", 

69 "NA", 

70 "NaN", 

71 "None", 

72 "n/a", 

73 "nan", 

74 "", 

75] 

76 

77_POTEC_TEXTS = [f"{domain}{i}" for domain in ("b", "p") for i in range(6)] 

78 

79#: DATA-65 — one OSF file version, pinned. ``osf.io/download/<id>`` with no 

80#: version serves whatever was uploaded last, and PoTeC's own files have been 

81#: replaced five times since 2023 (the scanpaths archive is at version 5, of 

82#: 2026-06-17), so an unpinned download could hand two users two corpora under 

83#: one name — and the dataset table's published figures (DATA-36) would describe 

84#: neither. A pin is the file's version *and* its size in bytes, which the 

85#: download is checked against. 

86_OSF_DOWNLOAD_URL = "https://osf.io/download/{resource}/?version={version}" 

87 

88 

89@dataclass(frozen=True) 

90class OsfFile: 

91 """One pinned OSF file: its storage id, version and size in bytes.""" 

92 

93 resource: str 

94 version: int 

95 size: int 

96 

97 @property 

98 def url(self) -> str: 

99 return _OSF_DOWNLOAD_URL.format(resource=self.resource, version=self.version) 

100 

101 

102def _check_download_size(got: int, pin: OsfFile, detail: str) -> None: 

103 """Refuse a download that is not the pinned file (DATA-65).""" 

104 if got != pin.size: 

105 raise OSError( 

106 f"{detail}: OSF sent a different file ({got:,} bytes, expected " 

107 f"{pin.size:,}). Try again later." 

108 ) 

109 

110 

111# OSF storage ids from the PoTeC repo's download_data_files.py, at the versions 

112# current on 2026-10-02. 

113_POTEC_OSF_RESOURCES = { 

114 "scanpaths": OsfFile("thgv2", 5, 9_796_046), 

115 "fixations": OsfFile("53zwb", 2, 6_456_542), 

116 "reading_measures": OsfFile("g5jds", 6, 3_326_244), 

117} 

118#: The AOI files come from the PoTeC repo itself — pinned to a commit (DATA-65), 

119#: not ``main``, for the same reason as the OSF versions above. 

120_POTEC_REPO_COMMIT = "46247ca2aea5876311acf6b59338880d0bf5449d" 

121_POTEC_RAW_URL = ( 

122 f"https://raw.githubusercontent.com/DiLi-Lab/PoTeC/{_POTEC_REPO_COMMIT}/{{path}}" 

123) 

124 

125 

126def _read_potec_tsv(path) -> pd.DataFrame: 

127 return pd.read_csv( 

128 path, sep="\t", keep_default_na=False, na_values=_POTEC_NA_VALUES 

129 ) 

130 

131 

132#: UX-168: download in reads of at most this size, reporting bytes after each — 

133#: the progress the card shows, and the checkpoint a Cancel stops at. Each is a 

134#: `read1`, which returns whatever has arrived: a `read` waits for the whole 

135#: MiB, so on a 100 KB/s line Stop took ~10 s to act. 

136_DOWNLOAD_CHUNK = 1 << 20 

137#: UX-168: seconds a download may wait on the network — to connect, or for its 

138#: next bytes — before giving up. Without it a stalled connection blocked its 

139#: read forever, so Stop, which acts between reads, never could; the timeout 

140#: surfaces as an `OSError`, which both ⬇ Download buttons already report. 

141_DOWNLOAD_TIMEOUT_S = 60 

142 

143 

144def _content_length(response) -> int | None: 

145 value = response.headers.get("Content-Length") 

146 return int(value) if value and str(value).isdigit() else None 

147 

148 

149def _read_body(response, write: Callable[[bytes], object], *, detail: str) -> None: 

150 """Hand ``response``'s body to ``write`` as it arrives, with progress. 

151 

152 Only a body that arrives whole returns (UX-168). `read1` returns ``b""`` on 

153 an early EOF — a server, proxy or load balancer closing the connection 

154 mid-body — exactly as at the real end, so a body shorter than its 

155 ``Content-Length`` is a `ConnectionError`; so is a chunked body cut short, 

156 which `http.client` raises as an `IncompleteRead` (an `HTTPException`, not 

157 an `OSError`). Both ⬇ Download buttons report an `OSError`, and a caller 

158 must never commit what was read. 

159 """ 

160 total = _content_length(response) 

161 done = 0 

162 try: 

163 while chunk := response.read1(_DOWNLOAD_CHUNK): 

164 write(chunk) 

165 done += len(chunk) 

166 progress.report(done, total, unit="bytes", detail=detail) 

167 except http.client.HTTPException as exc: 

168 raise ConnectionError( 

169 f"the download stopped after {done:,} bytes; try again" 

170 ) from exc 

171 if total is not None and done < total: 

172 raise ConnectionError( 

173 f"the download stopped at {done:,} of {total:,} bytes; try again" 

174 ) 

175 

176 

177def _open_url(url: str): 

178 """Open ``url`` for a download, trusting what the operating system trusts. 

179 

180 Python's own `ssl` defaults read OpenSSL's CA list, which a python.org 

181 install on macOS ships empty until *Install Certificates.command* is run, 

182 and which never holds the root a TLS-inspecting campus or company proxy 

183 re-signs with. Either way every download failed with 

184 ``CERTIFICATE_VERIFY_FAILED``. `truststore` verifies against the macOS 

185 Keychain / Windows certificate store / the system bundle instead, as the 

186 browser does. 

187 """ 

188 context = truststore.SSLContext(ssl.PROTOCOL_TLS_CLIENT) 

189 return urllib.request.urlopen(url, timeout=_DOWNLOAD_TIMEOUT_S, context=context) 

190 

191 

192def _fetch_bytes(url: str, *, detail: str) -> bytes: 

193 """``url``'s body, read as it arrives with progress (UX-168) — all of it, or 

194 a `ConnectionError` (`_read_body`).""" 

195 with _open_url(url) as response: 

196 buffer = io.BytesIO() 

197 _read_body(response, buffer.write, detail=detail) 

198 return buffer.getvalue() 

199 

200 

201def _fetch_to_file(url: str, dest: Path, *, detail: str) -> None: 

202 """Stream ``url`` into ``dest`` through a ``.part`` file (UX-168). 

203 

204 The ``.part`` → final rename keeps an interrupted fetch from passing for a 

205 complete file, and it happens only once the whole body has arrived 

206 (`_read_body`); a cancel (or any failure) deletes the partial file — and a 

207 cleanup that fails in turn never masks the error that caused it. 

208 """ 

209 tmp = dest.with_name(dest.name + ".part") 

210 try: 

211 with ( 

212 _open_url(url) as response, 

213 tmp.open("wb") as out, 

214 ): 

215 _read_body(response, out.write, detail=detail) 

216 tmp.replace(dest) 

217 except BaseException: 

218 with contextlib.suppress(OSError): 

219 tmp.unlink(missing_ok=True) 

220 raise 

221 

222 

223def download_potec(root, *, fixation_source: str = "scanpaths") -> Path: 

224 """Download the PoTeC files :func:`load_potec` needs into ``root``. 

225 

226 Fetches the per-trial eye-tracking archive (~10 MB zip) from PoTeC's OSF 

227 repository and the 24 per-text AOI files (word boxes + character boxes) 

228 from the PoTeC GitHub repo. Skips anything already present, so it's safe 

229 to call repeatedly (and it's a no-op on a full clone of the PoTeC repo 

230 where ``download_data_files.py`` has been run). 

231 

232 ``fixation_source`` is ``"scanpaths"`` (default; temporally ordered 

233 fixations with word indices) or ``"fixations"``. 

234 """ 

235 if fixation_source not in _POTEC_OSF_RESOURCES: 

236 raise ValueError( 

237 f"fixation_source must be one of {sorted(_POTEC_OSF_RESOURCES)}, " 

238 f"got {fixation_source!r}" 

239 ) 

240 root = Path(root) 

241 

242 eyetracking_dir = root / "eyetracking_data" / fixation_source 

243 if not eyetracking_dir.is_dir(): 

244 pin = _POTEC_OSF_RESOURCES[fixation_source] 

245 print(f"Downloading PoTeC {fixation_source} from {pin.url} …") 

246 detail = f"PoTeC {fixation_source} archive" 

247 payload = _fetch_bytes(pin.url, detail=detail) 

248 _check_download_size(len(payload), pin, detail) 

249 target = root / "eyetracking_data" 

250 target.mkdir(parents=True, exist_ok=True) 

251 # UX-168: unpack into a staging folder and rename it into place only 

252 # when complete. A cancel mid-unpack would otherwise leave a partial 

253 # `scanpaths/` that `potec_present`'s "any .tsv" check accepts. 

254 staging = target / f".{fixation_source}.part" 

255 shutil.rmtree(staging, ignore_errors=True) 

256 try: 

257 with zipfile.ZipFile(io.BytesIO(payload)) as archive: 

258 members = [ 

259 m 

260 for m in archive.namelist() 

261 # The OSF zips carry macOS resource-fork cruft; keep only the 

262 # real per-trial TSVs. 

263 if m.startswith(f"{fixation_source}/") and m.endswith(".tsv") 

264 ] 

265 if not members: 

266 # Say what is wrong, not that the staging folder is missing. 

267 raise ValueError( 

268 f"The PoTeC archive from {pin.url} holds no " 

269 f"{fixation_source}/*.tsv files; its layout may have changed." 

270 ) 

271 for index, member in enumerate(members, start=1): 

272 archive.extract(member, staging) 

273 progress.report( 

274 index, 

275 len(members), 

276 unit="files", 

277 detail="Unpacking the archive", 

278 ) 

279 (staging / fixation_source).replace(eyetracking_dir) 

280 except zipfile.BadZipFile as exc: 

281 # UX-168: a damaged archive — opened or extracted — is a data error 

282 # both ⬇ Download buttons report, not a raw traceback. 

283 raise ValueError( 

284 "The downloaded PoTeC archive is damaged; download it again." 

285 ) from exc 

286 finally: 

287 shutil.rmtree(staging, ignore_errors=True) 

288 

289 for text_id in _POTEC_TEXTS: 

290 for rel in ( 

291 f"stimuli/word_aoi_texts/word_aoi_{text_id}.tsv", 

292 f"stimuli/aoi_texts/{text_id}.ias", 

293 ): 

294 dest = root / rel 

295 if dest.is_file(): 

296 continue 

297 dest.parent.mkdir(parents=True, exist_ok=True) 

298 url = _POTEC_RAW_URL.format(path=rel) 

299 print(f"Downloading {url} …") 

300 # Write to a temp file and atomically rename into place, so an 

301 # interrupted fetch never leaves a truncated AOI file that 

302 # `dest.is_file()` / `potec_present` would then treat as complete — 

303 # mirroring download_onestop. 

304 _fetch_to_file(url, dest, detail=f"AOI file {rel.rsplit('/', 1)[-1]}") 

305 return root 

306 

307 

308def potec_present(root) -> bool: 

309 """True when ``root`` already holds the full PoTeC corpus a load needs. 

310 

311 The app loads *every* text, so this requires the per-text word + character 

312 AOI files for **all** texts (exactly what :func:`download_potec` fetches), 

313 plus a populated eye-tracking folder. A lenient "any AOI file" check would 

314 pass a partial tree and then crash mid-load (`_potec_words` raises on the 

315 first missing text) with no way to recover; requiring all of them instead 

316 lets the app offer the Download button, which self-heals the gap. Cheap 

317 (path stats only), so the status shows without reading any CSV.""" 

318 root = Path(root) 

319 base = root / "eyetracking_data" 

320 has_fixations = any( 

321 (base / s).is_dir() and any((base / s).glob("*.tsv")) 

322 for s in ("scanpaths", "fixations") 

323 ) 

324 word_dir = root / "stimuli" / "word_aoi_texts" 

325 char_dir = root / "stimuli" / "aoi_texts" 

326 has_all_aoi = all( 

327 (word_dir / f"word_aoi_{text_id}.tsv").is_file() 

328 and (char_dir / f"{text_id}.ias").is_file() 

329 for text_id in _POTEC_TEXTS 

330 ) 

331 return has_fixations and has_all_aoi 

332 

333 

334def _potec_words(root: Path, texts: Iterable[str]) -> pd.DataFrame: 

335 """Stimulus-level word table: one row per word per text, with boxes. 

336 

337 PoTeC keys the text id in the AOI *filename* only; it becomes a regular 

338 ``text_id`` column here. Line indices come from the character-level 

339 ``.ias`` files (the word AOI files don't carry them) via the lines' y 

340 positions.""" 

341 frames = [] 

342 for text_id in texts: 

343 path = root / "stimuli" / "word_aoi_texts" / f"word_aoi_{text_id}.tsv" 

344 if not path.is_file(): 

345 raise FileNotFoundError( 

346 f"PoTeC word AOI file not found: {path} — pass download=True " 

347 "or run PoTeC's download_data_files.py in a repo clone." 

348 ) 

349 words = _read_potec_tsv(path) 

350 words["text_id"] = text_id 

351 

352 ias = _read_potec_ias(root, text_id) 

353 # Character boxes on the same text line share start_y, so the char 

354 # AOIs give an exact y → line lookup for the word boxes. 

355 y_to_line = ias.drop_duplicates("start_y").set_index("start_y")["line"] 

356 words["line"] = words["start_y"].map(y_to_line) 

357 frames.append(words) 

358 return pd.concat(frames, ignore_index=True) 

359 

360 

361def _read_potec_ias(root: Path, text_id: str) -> pd.DataFrame: 

362 path = root / "stimuli" / "aoi_texts" / f"{text_id}.ias" 

363 if not path.is_file(): 

364 raise FileNotFoundError( 

365 f"PoTeC character AOI file not found: {path} — pass download=True " 

366 "or run PoTeC's download_data_files.py in a repo clone." 

367 ) 

368 return _read_potec_tsv(path) 

369 

370 

371def _potec_fixations( 

372 root: Path, 

373 texts: Iterable[str], 

374 readers: Iterable | None = None, 

375) -> pd.DataFrame: 

376 """Concatenated per-trial fixation files with reconstructed coordinates. 

377 

378 Prefers ``eyetracking_data/scanpaths/`` (fixations in temporal order, with 

379 word indices) and falls back to ``eyetracking_data/fixations/``. Each 

380 fixation's (x, y) is the center of the fixated character's box from the 

381 per-text ``.ias`` file — PoTeC discards the original screen coordinates.""" 

382 base = root / "eyetracking_data" 

383 source = next((s for s in ("scanpaths", "fixations") if (base / s).is_dir()), None) 

384 if source is None: 

385 raise FileNotFoundError( 

386 f"No PoTeC fixation data under {base} — expected a 'scanpaths' or " 

387 "'fixations' folder. Pass download=True, or run PoTeC's " 

388 "download_data_files.py in a repo clone." 

389 ) 

390 suffix = "scanpath" if source == "scanpaths" else "fixations" 

391 

392 reader_set = None if readers is None else {str(r) for r in readers} 

393 # UX-166: gather the files first, so the card can say "312 of 900 files". 

394 jobs: list[tuple[Path, pd.DataFrame]] = [] 

395 for text_id in texts: 

396 char_boxes = _read_potec_ias(root, text_id) 

397 char_x = (char_boxes["start_x"] + char_boxes["end_x"]) / 2.0 

398 char_y = (char_boxes["start_y"] + char_boxes["end_y"]) / 2.0 

399 centers = pd.DataFrame( 

400 {"aoi": char_boxes["aoi"], "x": char_x, "y": char_y} 

401 ).drop_duplicates("aoi") 

402 for path in sorted((base / source).glob(f"reader*_{text_id}_{suffix}.tsv")): 

403 reader_id = path.stem.removeprefix("reader").split("_")[0] 

404 if reader_set is not None and reader_id not in reader_set: 

405 continue 

406 jobs.append((path, centers)) 

407 # UX-166: "0 of N" before the first file, so a gated card is armed from the 

408 # start — a report only after each file would hide the first one. 

409 progress.report(0, len(jobs), unit="files") 

410 frames = [] 

411 for index, (path, centers) in enumerate(jobs, start=1): 

412 fixations = _read_potec_tsv(path) 

413 frames.append(fixations.merge(centers, on="aoi", how="left")) 

414 progress.report(index, len(jobs), unit="files") 

415 if not frames: 

416 raise FileNotFoundError( 

417 f"No PoTeC fixation files matched the requested readers/texts " 

418 f"under {base / source}." 

419 ) 

420 return pd.concat(frames, ignore_index=True, sort=False) 

421 

422 

423# Column mappings from the raw PoTeC frames to the canonical schema. Explicit 

424# (rather than relying on auto-detection) so the loader stays stable even if 

425# PoTeC adds columns. No participant on words: the word boxes are 

426# stimulus-level and get broadcast across readers. Shared by load_potec and 

427# the app's PoTeC data source, which declares them over its auto-detection. 

428# 

429# A trial is one reader's reading of one text, so the fixations' Trial ID is 

430# the composite ``reader_id`` + ``text_id`` (``"0_b0"``): the text name alone 

431# repeats across all 75 readers. The word boxes stay keyed by text, and each 

432# reading finds its text's boxes through the Text ID both tables map (DATA-49). 

433POTEC_WORD_SCHEMA = dict( 

434 participant=None, 

435 trial="text_id", 

436 text_id="text_id", 

437 word_id="aoi", 

438 text="word", 

439 line="line", 

440 left="start_x", 

441 right="end_x", 

442 top="start_y", 

443 bottom="end_y", 

444) 

445POTEC_FIX_SCHEMA = dict( 

446 participant="reader_id", 

447 trial=["reader_id", "text_id"], 

448 text_id="text_id", 

449 duration="fixation_duration", 

450 x="x", 

451 y="y", 

452 fixation_id="fixation_index", 

453 word_id="word_index_in_text", 

454) 

455 

456# PoTeC presentation monitor (DELL P2210, 60 Hz). Pass as ``canvas_size`` to 

457# plot_scanpath for true-to-scale rendering. 

458POTEC_MONITOR = (1680, 1050) 

459 

460 

461def potec_raw_frames( 

462 root, 

463 *, 

464 readers: Iterable | None = None, 

465 texts: Iterable[str] | None = None, 

466 download: bool = False, 

467) -> tuple[pd.DataFrame, pd.DataFrame]: 

468 """Raw (pre-normalization) PoTeC ``(words, fixations)`` frames. 

469 

470 Same inputs as :func:`load_potec`, but returns the frames *before* schema 

471 normalization — for callers that run their own auto-detection / column 

472 mapping (e.g. the Streamlit app's PoTeC data source). Word boxes are 

473 stimulus-level (one row per word per text, ``text_id`` column, no 

474 participant); fixations carry reconstructed ``x``/``y`` from the fixated 

475 character's box center. Use :func:`load_potec` for the normalized, 

476 ready-to-plot frames. 

477 """ 

478 root = Path(root) 

479 if download: 

480 download_potec(root) 

481 texts = list(texts) if texts is not None else list(_POTEC_TEXTS) 

482 unknown = sorted(set(texts) - set(_POTEC_TEXTS)) 

483 if unknown: 

484 raise ValueError(f"Unknown PoTeC text ids: {unknown} (valid: {_POTEC_TEXTS})") 

485 return _potec_words(root, texts), _potec_fixations(root, texts, readers) 

486 

487 

488def load_potec( 

489 root, 

490 *, 

491 readers: Iterable | None = None, 

492 texts: Iterable[str] | None = None, 

493 download: bool = False, 

494 names: str = "source", 

495) -> tuple[pd.DataFrame, pd.DataFrame]: 

496 """Load PoTeC as normalized ``(words, fixations)`` frames, ready to plot. 

497 

498 Under PoTeC's own column names; ``names="canonical"`` for the internal 

499 ones (see `api.load_scanpath_data`). 

500 

501 ``root`` is a clone of the PoTeC repo (with the eye-tracking data 

502 downloaded) or any folder; with ``download=True`` the needed files are 

503 fetched into it on first use (~10 MB). Narrow the load with ``readers`` 

504 (e.g. ``[0, 1]``) and/or ``texts`` (e.g. ``["b0", "p3"]``) — the full 

505 corpus is 75 readers × 12 texts = 900 trials. 

506 

507 Participants are PoTeC reader ids (as strings); a trial is one reader's 

508 reading of one text, ``<reader>_<text>`` (``"0_b0"``), and ``text_id`` is 

509 the text (``b0``–``b5`` biology, ``p0``–``p5`` physics):: 

510 

511 import scanpath_studio as sps 

512 

513 words, fixations = sps.load_potec("data/PoTeC", readers=[0], texts=["b0"]) 

514 fig = sps.plot_scanpath( 

515 words, fixations, "0", "0_b0", canvas_size=(1680, 1050) 

516 ) 

517 

518 The PoTeC monitor was 1680×1050 (DELL P2210, 60 Hz); pass that as ``canvas_size`` to 

519 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] for true-to-scale rendering. 

520 """ 

521 words_raw, fixations_raw = potec_raw_frames( 

522 root, readers=readers, texts=texts, download=download 

523 ) 

524 

525 from . import api 

526 

527 return api.load_scanpath_data( 

528 words=words_raw, 

529 fixations=fixations_raw, 

530 word_schema=dict(POTEC_WORD_SCHEMA), 

531 fix_schema=dict( 

532 POTEC_FIX_SCHEMA, 

533 word_id=( 

534 "word_index_in_text" 

535 if "word_index_in_text" in fixations_raw.columns 

536 else None 

537 ), 

538 ), 

539 names=names, 

540 ) 

541 

542 

543# --------------------------------------------------------------------------- 

544# OneStop Eye Movements — 360-participant English corpus (Berzak et al. 2025, 

545# https://github.com/lacclab/OneStop-Eye-Movements). Distributed on OSF as 

546# interest-area (word) + fixation reports, split by reading **regime** and by 

547# trial **part** (which screen of a trial — title / question preview / 

548# paragraph / questions / answers / QA / feedback). The reports share the 

549# bundled demo's schema (the demo is a 3-pid subset of the Paragraph part), so 

550# the generic auto-detect → normalize pipeline handles every part with no 

551# part-specific column mapping — this loader only fetches, reads, and (when more 

552# than one part is loaded) folds the part into the trial identity so the parts 

553# don't collide. 

554# 

555# Two **variants**: 

556# * ``public`` — the OSF download-on-demand release (this module fetches it). 

557# * ``lacclab`` — a lab-processed local export with ~40 extra derived columns 

558# (``unique_paragraph_id``, span indices, normalized dwell, …); a superset 

559# of the public schema, so it flows through the same pipeline (and its 

560# ``unique_paragraph_id`` wins in normalization). No download — a local path. 

561# 

562# Distinct from the env-var "OneStop server bundle" source 

563# (``data.load_onestop_server_bundle``), which serves a lacclab export via 

564# ``$ONESTOP_DATA_DIR`` and its per-pid shards for review-app deep links. 

565# --------------------------------------------------------------------------- 

566 

567#: DATA-65 — every OneStop report is at OSF version 1 (2025-05-28, checked 

568#: 2026-10-02); id → size in bytes, the pin `download_onestop` checks against. 

569_ONESTOP_OSF_VERSION = 1 

570_ONESTOP_OSF_SIZES = { 

571 # all-regimes full release, interest areas 

572 "u7f9b": 8_287_933, 

573 "zn473": 23_885_174, 

574 "zhywq": 397_501_066, 

575 "tcv9h": 43_745_513, 

576 "q3shp": 130_794_114, 

577 "3j8av": 151_191_442, 

578 "t6n8v": 12_472_761, 

579 # all-regimes full release, fixations 

580 "uwz2e": 17_724_147, 

581 "7a3md": 41_597_351, 

582 "tbxdc": 597_284_698, 

583 "cmx6k": 64_566_806, 

584 "ax4md": 185_658_405, 

585 "fg7se": 207_300_032, 

586 "e76vz": 25_763_786, 

587 # per-regime Paragraph reports 

588 "xkgfz": 177_291_322, 

589 "ne4az": 288_349_184, 

590 "yxzte": 165_756_770, 

591 "bznfk": 245_787_943, 

592 "dwfk4": 28_875_267, 

593 "83ctd": 36_529_359, 

594 "ygjup": 25_606_136, 

595 "paqn8": 26_657_407, 

596} 

597 

598# The seven trial parts (interest periods), in presentation order. Each maps to 

599# one interest-area + one fixation OSF report in the ``onestop-full`` release 

600# (all-regimes). Paragraph is the reading passage; the others are the surrounding 

601# screens (title, the pre/post question, its four answers, the correctness 

602# feedback) — except QA, which is not a screen: it is the Questions and Answers 

603# interest periods taken together (the corpus gives its page as "3+4"), so its 

604# fixations are theirs again (DATA-64). All share the Paragraph report schema 

605# (IA_LEFT/RIGHT/TOP/BOTTOM boxes, IA_LABEL word text, per-word reading measures), 

606# so every part renders as a scanpath. Keep the display order = presentation order. 

607_ONESTOP_PARTS: tuple[str, ...] = ( 

608 "Title", 

609 "Question_Preview", 

610 "Paragraph", 

611 "Questions", 

612 "Answers", 

613 "QA", 

614 "Feedback", 

615) 

616ONESTOP_DEFAULT_PARTS: tuple[str, ...] = ("Paragraph",) 

617 

618# OSF ids for the all-regimes **full** release (every part), from the OneStop 

619# repo's download_data_files.py "onestop-full" group. kind → part → OSF id. 

620_ONESTOP_FULL_OSF = { 

621 "ia": { 

622 "Title": "u7f9b", 

623 "Question_Preview": "zn473", 

624 "Paragraph": "zhywq", 

625 "Questions": "tcv9h", 

626 "Answers": "q3shp", 

627 "QA": "3j8av", 

628 "Feedback": "t6n8v", 

629 }, 

630 "fixations": { 

631 "Title": "uwz2e", 

632 "Question_Preview": "7a3md", 

633 "Paragraph": "tbxdc", 

634 "Questions": "cmx6k", 

635 "Answers": "ax4md", 

636 "QA": "fg7se", 

637 "Feedback": "e76vz", 

638 }, 

639} 

640 

641# OSF ids for the per-regime **Paragraph-only** releases (the four reading 

642# regimes each ship just the paragraph reports, filtered to that regime), from 

643# the "ordinary"/"information_seeking"/"repeated"/"information_seeking_repeated" 

644# groups. regime → kind → OSF id. Only the Paragraph part is regime-split on OSF; 

645# the other parts come from the all-regimes full release (see _ONESTOP_FULL_OSF). 

646_ONESTOP_REGIMES = { 

647 "ordinary": dict(ia="xkgfz", fixations="ne4az"), 

648 "information_seeking": dict(ia="yxzte", fixations="bznfk"), 

649 "repeated": dict(ia="dwfk4", fixations="83ctd"), 

650 "information_seeking_repeated": dict(ia="ygjup", fixations="paqn8"), 

651} 

652 

653ONESTOP_VARIANTS = ("public", "lacclab") 

654 

655# DATA-63: what makes a trial belong to a regime, as the corpus records it on 

656# every report — ``(question_preview, repeated_reading_trial)``. The per-regime 

657# Paragraph files are exactly these slices (checked against the OSF ordinary 

658# file: every row is ``(False, False)``), and the all-regimes reports of the 

659# other parts carry the same two columns, so they are cut to the regime by the 

660# same rule instead of handing every regime's trials to each one. 

661_ONESTOP_REGIME_FLAGS: dict[str, tuple[bool, bool]] = { 

662 "ordinary": (False, False), 

663 "information_seeking": (True, False), 

664 "repeated": (False, True), 

665 "information_seeking_repeated": (True, True), 

666} 

667_ONESTOP_REGIME_COLUMNS: tuple[str, ...] = ( 

668 "question_preview", 

669 "repeated_reading_trial", 

670) 

671 

672 

673def onestop_regime_parts(regime: str) -> list[str]: 

674 """Every screen a reader in ``regime`` saw, in presentation order. 

675 

676 The parts other than QA, and of those the question-preview screen only for 

677 the information-seeking regimes — its report holds no trial of the others, 

678 so loading it there would download a file to keep none of it. QA is left 

679 out (DATA-64): it repeats the Questions and Answers screens' fixations as 

680 one interest period, so loading it beside them counts each fixation twice 

681 and shows that stretch of the trial as a further screen. ``parts=["QA"]`` 

682 still loads it on request. 

683 """ 

684 if regime not in _ONESTOP_REGIME_FLAGS: 

685 raise ValueError( 

686 f"regime must be one of {sorted(_ONESTOP_REGIME_FLAGS)}, got {regime!r}" 

687 ) 

688 preview, _ = _ONESTOP_REGIME_FLAGS[regime] 

689 return [ 

690 p for p in _ONESTOP_PARTS if p != "QA" and (preview or p != "Question_Preview") 

691 ] 

692 

693 

694def _keep_onestop_regime(frame: pd.DataFrame, regime: str) -> pd.DataFrame: 

695 """The rows of an all-regimes report that belong to ``regime``. 

696 

697 A frame without the two flag columns is returned untouched: there is 

698 nothing to tell its regimes apart by. 

699 """ 

700 from . import data 

701 

702 if not set(_ONESTOP_REGIME_COLUMNS) <= set(frame.columns): 

703 return frame 

704 preview, repeated = _ONESTOP_REGIME_FLAGS[regime] 

705 mask = (data.coerce_flag(frame["question_preview"]) == preview) & ( 

706 data.coerce_flag(frame["repeated_reading_trial"]) == repeated 

707 ) 

708 return frame.loc[mask].reset_index(drop=True) 

709 

710 

711def _onestop_osf_resource(kind: str, part: str, regime: str) -> str | None: 

712 """OSF id for a (kind, part, regime), or None when not published. 

713 

714 Paragraph is regime-split (four separate downloads); every other part only 

715 exists in the all-regimes full release, so it uses the full-release id 

716 regardless of regime.""" 

717 if part == "Paragraph" and regime in _ONESTOP_REGIMES: 

718 return _ONESTOP_REGIMES[regime].get(kind) 

719 return _ONESTOP_FULL_OSF.get(kind, {}).get(part) 

720 

721 

722def _onestop_report_path( 

723 root: Path, kind: str, regime: str, part: str = "Paragraph" 

724) -> Path: 

725 """Local path of a public-variant OneStop report CSV.zip. 

726 

727 Paragraph keeps the historical ``<kind>_Paragraph_<regime>.csv.zip`` name 

728 (per-regime download). Other parts are all-regimes, so they use 

729 ``<kind>_<Part>.csv.zip`` (matching the OSF full-release filenames).""" 

730 if part == "Paragraph": 

731 return root / f"{kind}_Paragraph_{regime}.csv.zip" 

732 return root / f"{kind}_{part}.csv.zip" 

733 

734 

735def _onestop_lacclab_report_path(root: Path, kind: str, part: str) -> Path: 

736 """Local path of a lacclab-variant OneStop report CSV.zip. 

737 

738 The lacclab export names files plainly by part (no regime suffix) — e.g. 

739 ``ia_Paragraph.csv.zip`` / ``fixations_Paragraph.csv.zip`` — since a lacclab 

740 export folder holds one regime's reports.""" 

741 return root / f"{kind}_{part}.csv.zip" 

742 

743 

744def _onestop_part_paths( 

745 root: Path, kind: str, regime: str, part: str, variant: str 

746) -> Path: 

747 """Dispatch to the public or lacclab path convention for one report.""" 

748 if variant == "lacclab": 

749 return _onestop_lacclab_report_path(root, kind, part) 

750 return _onestop_report_path(root, kind, regime, part) 

751 

752 

753def _normalize_onestop_parts(parts: Iterable[str] | None) -> list: 

754 """Validate + order a requested parts selection (defaults to Paragraph).""" 

755 if not parts: 

756 return list(ONESTOP_DEFAULT_PARTS) 

757 requested = {str(p) for p in parts} 

758 unknown = sorted(requested - set(_ONESTOP_PARTS)) 

759 if unknown: 

760 raise ValueError( 

761 f"Unknown OneStop parts: {unknown} (valid: {list(_ONESTOP_PARTS)})" 

762 ) 

763 # Keep presentation order regardless of how they were passed. 

764 return [p for p in _ONESTOP_PARTS if p in requested] 

765 

766 

767def onestop_present( 

768 root, 

769 *, 

770 regime: str = "ordinary", 

771 parts: Iterable[str] | None = None, 

772 variant: str = "public", 

773) -> bool: 

774 """True when ``root`` holds the IA + fixation reports for every chosen part. 

775 

776 Lets the app show a *found vs. download* status before any (large) read. 

777 ``variant`` selects the file-name convention (public OSF vs lacclab local).""" 

778 root = Path(root) 

779 for part in _normalize_onestop_parts(parts): 

780 for kind in ("ia", "fixations"): 

781 if not _onestop_part_paths(root, kind, regime, part, variant).is_file(): 

782 return False 

783 return True 

784 

785 

786def download_onestop( 

787 root, 

788 *, 

789 regime: str = "ordinary", 

790 parts: Iterable[str] | None = None, 

791) -> Path: 

792 """Download a OneStop regime + parts' IA + fixation reports into ``root``. 

793 

794 Fetches the two CSV.zip reports for each chosen ``part`` from OneStop's OSF 

795 release into ``root``, skipping any already present, so it's safe to call 

796 repeatedly. The reports are large (tens to hundreds of MB each); caching them 

797 on disk means only the first load pays the download. 

798 

799 ``regime`` is one of ``ordinary``, ``information_seeking``, ``repeated``, 

800 ``information_seeking_repeated``. ``parts`` is any subset of 

801 ``Title / Question_Preview / Paragraph / Questions / Answers / QA / 

802 Feedback`` (default: just Paragraph). Only Paragraph is regime-split on OSF; 

803 the other parts come from the all-regimes full release. 

804 

805 (Download is a *public*-variant operation — the lacclab variant is a local 

806 export with no download URL.) 

807 """ 

808 if regime not in _ONESTOP_REGIMES: 

809 raise ValueError( 

810 f"regime must be one of {sorted(_ONESTOP_REGIMES)}, got {regime!r}" 

811 ) 

812 root = Path(root) 

813 root.mkdir(parents=True, exist_ok=True) 

814 for part in _normalize_onestop_parts(parts): 

815 for kind in ("ia", "fixations"): 

816 dest = _onestop_report_path(root, kind, regime, part) 

817 if dest.is_file(): 

818 continue 

819 resource = _onestop_osf_resource(kind, part, regime) 

820 if resource is None: 

821 continue 

822 pin = OsfFile(resource, _ONESTOP_OSF_VERSION, _ONESTOP_OSF_SIZES[resource]) 

823 print(f"Downloading OneStop {regime} {part} {kind} report from {pin.url} …") 

824 # Write to a temp file and atomically rename into place, so an 

825 # interrupted write (killed process / full disk) never leaves a 

826 # truncated .csv.zip that `dest.is_file()` would then skip forever — 

827 # forcing a manual delete. The reports are large, so the window is real. 

828 detail = f"{part} {kind} report" 

829 _fetch_to_file(pin.url, dest, detail=detail) 

830 try: 

831 _check_download_size(dest.stat().st_size, pin, detail) 

832 except OSError: 

833 dest.unlink(missing_ok=True) 

834 raise 

835 return root 

836 

837 

838# BUG-43: the public OSF release ships **no** ``unique_paragraph_id``, so the 

839# only text identity on its reports is ``paragraph_id`` — the paragraph's index 

840# *within its article*, 1..7. On its own that makes paragraph 3 of article 1 and 

841# paragraph 3 of article 27 the same "text" (162 paragraphs read as 7), and 

842# because the same id then repeats dozens of times per reader, 

843# ``data._disambiguate_repeated_readings`` ranks the collisions apart into trial 

844# ids like ``3_r17`` — an order, not an identity. 

845# 

846# So compose the ids the corpus's own tooling writes, in the same field order 

847# ``update_sample_data.add_unique_ids`` uses for the bundled demo. The demo is a 

848# subset of this corpus, so a trial must not change identity depending on which 

849# of the two was opened. The level is part of the *text* id there, and stays 

850# one here: OneStop's Advanced and Elementary renderings of a paragraph are 

851# different texts, not two views of one. 

852_ONESTOP_PARAGRAPH_ID_PARTS: tuple[str, ...] = ( 

853 "article_batch", 

854 "article_id", 

855 "paragraph_id", 

856 "difficulty_level", 

857) 

858_ONESTOP_TRIAL_ID_PARTS: tuple[str, ...] = ("participant_id", "repeated_reading_trial") 

859#: Every source column the composition needs — handed to the planned read so 

860#: narrowing it (PERF-6) can never quietly cost us the ids. 

861_ONESTOP_ID_COLUMNS: tuple[str, ...] = ( 

862 _ONESTOP_PARAGRAPH_ID_PARTS + _ONESTOP_TRIAL_ID_PARTS 

863) 

864 

865 

866def _compose_onestop_ids(frame: pd.DataFrame) -> pd.DataFrame: 

867 """Give an OneStop report the ``unique_paragraph_id`` / ``unique_trial_id`` 

868 the public release omits (BUG-43). 

869 

870 Each id is composed only when the frame does not already carry it: the 

871 lacclab export ships its own ``unique_paragraph_id`` (in its own field 

872 order), and that is the publisher's identity — the trial id is composed 

873 *under* it rather than over it. A frame missing any of the composing 

874 columns is returned untouched: there is nothing to invent an id from. 

875 """ 

876 from . import data 

877 

878 columns = set(frame.columns) 

879 if "unique_paragraph_id" not in columns: 

880 if not set(_ONESTOP_PARAGRAPH_ID_PARTS) <= columns: 

881 return frame 

882 frame = frame.copy() 

883 frame["unique_paragraph_id"] = _join_ids(frame, _ONESTOP_PARAGRAPH_ID_PARTS) 

884 columns = set(frame.columns) 

885 if "unique_trial_id" not in columns: 

886 if not set(_ONESTOP_TRIAL_ID_PARTS) <= columns: 

887 return frame 

888 frame = frame.copy() 

889 reading = data.coerce_flag(frame["repeated_reading_trial"]).astype(int) 

890 frame["unique_trial_id"] = ( 

891 frame["participant_id"].astype(str) 

892 + "_" 

893 + frame["unique_paragraph_id"].astype(str) 

894 + "_r" 

895 + reading.astype(str) 

896 ) 

897 return frame 

898 

899 

900def _join_ids(frame: pd.DataFrame, columns: Iterable[str]) -> pd.Series: 

901 """``"_"``-joined string form of ``columns`` — one composed id per row. 

902 

903 Each part goes through ``stable_id`` (BUG-44), not a plain ``.astype(str)`` 

904 — several of OneStop's own id components (``paragraph_id``, ``article_id``) 

905 are whole numbers, and one blank cell anywhere else in that CSV column is 

906 enough for pandas to read the whole thing as ``float64``. Left uncorrected 

907 here, the composed id would embed a ``.0`` that a plain string column on 

908 the other side of a join never has. 

909 """ 

910 from . import data 

911 

912 parts = [data.stable_id(frame[column]) for column in columns] 

913 joined = parts[0] 

914 for part in parts[1:]: 

915 joined = joined + "_" + part 

916 return joined 

917 

918 

919def _read_onestop_part( 

920 root: Path, kind: str, regime: str, part: str, variant: str 

921) -> pd.DataFrame: 

922 """Read one part's report and stamp a ``part`` column onto it. 

923 

924 QA repeats the question words in the answer region with the *same* IA_ID, so 

925 its per-trial word ids aren't unique — drop the exact-duplicate rows so the 

926 fixation→word assignment (which keys on word id) keeps one box per word.""" 

927 from . import data 

928 

929 path = _onestop_part_paths(root, kind, regime, part, variant) 

930 if not path.is_file(): 

931 raise FileNotFoundError( 

932 f"OneStop {regime} {part} {kind} report not found: {path} — pass " 

933 "download=True to fetch it from OSF (public variant), or point at a " 

934 "folder holding the lacclab reports." 

935 ) 

936 # Read via data.read_mapped_table (not pd.read_csv directly): the OSF 

937 # .csv.zip archives wrap the CSV alongside macOS __MACOSX resource-fork 

938 # entries, which pandas' zip reader rejects ("Multiple files found in ZIP"). 

939 # read_table's zip path filters that cruft and reads with low_memory=False — 

940 # and PERF-6's planned read parses only the columns normalization keeps, 

941 # which is most of what a multi-gigabyte report costs. 

942 frame = data.read_mapped_table( 

943 path, 

944 kind="words" if kind == "ia" else "fixations", 

945 filter_fields=_ONESTOP_ID_COLUMNS + _ONESTOP_REGIME_COLUMNS, 

946 ) 

947 # DATA-63: only Paragraph is regime-split on OSF; every other public part 

948 # holds all four regimes' trials, so cut it to the one asked for. 

949 if variant == "public" and part != "Paragraph": 

950 frame = _keep_onestop_regime(frame, regime) 

951 frame = _compose_onestop_ids(frame) 

952 frame["part"] = part 

953 if part == "QA": 

954 frame = frame.drop_duplicates() 

955 return frame 

956 

957 

958def _onestop_parts_as_screens( 

959 words: pd.DataFrame, fixations: pd.DataFrame 

960) -> tuple[pd.DataFrame, pd.DataFrame]: 

961 """Make each part a *screen* of its trial, in presentation order (DATA-63). 

962 

963 Every part of a reading shares its ``unique_trial_id`` — the title, the 

964 passage and the question screens are one trial, shown one after another — 

965 so the trial is left whole and the part is its screen: the ``part`` column 

966 is auto-detected as ``screen_id`` (`data.SCREEN_ID_CANDIDATES`), and 

967 ``screen_index`` numbers the screens a trial *has* 1..N in `_ONESTOP_PARTS` 

968 order. Per trial, because the Screen picker reads it as "n of N": not every 

969 reading has every part (only an article's first paragraph has a title 

970 screen, only the information-seeking regimes a question preview), and a 

971 fixed per-part number would read "3 of 6" on the second screen. Words and 

972 fixations are numbered from the screens both hold, so the two agree. Each 

973 screen keeps its own coordinate space and word boxes (`multipart.py`). A 

974 frame without the ``part`` / ``unique_trial_id`` columns is returned 

975 untouched. 

976 """ 

977 keys = ["participant_id", "unique_trial_id"] 

978 needed = {*keys, "part"} 

979 if words.empty or not needed <= set(words.columns): 

980 return words, fixations 

981 order = {part: index for index, part in enumerate(_ONESTOP_PARTS)} 

982 screens = words[[*keys, "part"]].drop_duplicates().astype(str) 

983 screens["_order"] = screens["part"].map(order) 

984 screens = screens.sort_values([*keys, "_order"], kind="stable") 

985 screens["screen_index"] = screens.groupby(keys, sort=False).cumcount() + 1 

986 screens = screens.drop(columns="_order") 

987 

988 def _stamp(frame: pd.DataFrame) -> pd.DataFrame: 

989 if frame.empty or not needed <= set(frame.columns): 

990 return frame 

991 probe = frame[[*keys, "part"]].astype(str) 

992 index = probe.merge(screens, on=[*keys, "part"], how="left")["screen_index"] 

993 frame = frame.copy() 

994 frame["screen_index"] = index.to_numpy() 

995 return frame 

996 

997 return _stamp(words), _stamp(fixations) 

998 

999 

1000def onestop_raw_frames( 

1001 root, 

1002 *, 

1003 regime: str = "ordinary", 

1004 parts: Iterable[str] | None = None, 

1005 variant: str = "public", 

1006 download: bool = False, 

1007) -> tuple[pd.DataFrame, pd.DataFrame]: 

1008 """Raw (pre-normalization) OneStop ``(words, fixations)`` frames. 

1009 

1010 Reads the interest-area + fixation reports for each chosen ``part`` of 

1011 ``regime`` from ``root`` (fetching the public reports from OSF first when 

1012 ``download=True``). The reports already match the bundled demo's schema, so 

1013 the returned frames go through the same auto-detect → normalize path as an 

1014 upload — no OneStop-specific column mapping is needed here. 

1015 

1016 ``parts`` is any subset of the seven trial parts (default: Paragraph; 

1017 :func:`onestop_regime_parts` lists every part of a regime). A public part 

1018 other than Paragraph is cut to ``regime`` by its ``question_preview`` / 

1019 ``repeated_reading_trial`` flags (DATA-63). Each part is a *screen* of its 

1020 trial (``part`` → ``screen_id``, plus a ``screen_index`` in presentation 

1021 order), so a reading's title, passage and question screens are one trial. 

1022 ``variant`` is 

1023 ``"public"`` (OSF release) or ``"lacclab"`` (a local lab-processed export; 

1024 superset schema, no download). 

1025 """ 

1026 if regime not in _ONESTOP_REGIMES: 

1027 raise ValueError( 

1028 f"regime must be one of {sorted(_ONESTOP_REGIMES)}, got {regime!r}" 

1029 ) 

1030 if variant not in ONESTOP_VARIANTS: 

1031 raise ValueError( 

1032 f"variant must be one of {list(ONESTOP_VARIANTS)}, got {variant!r}" 

1033 ) 

1034 part_list = _normalize_onestop_parts(parts) 

1035 root = Path(root) 

1036 if download and variant == "public": 

1037 download_onestop(root, regime=regime, parts=part_list) 

1038 

1039 reports = [(kind, part) for kind in ("ia", "fixations") for part in part_list] 

1040 # UX-166: "0 of N" first — one report here is a whole IA or fixation file, 

1041 # so a report only after it would hide most of the load from a gated card. 

1042 progress.report(0, len(reports), unit="reports") 

1043 word_frames, fix_frames = [], [] 

1044 for index, (kind, part) in enumerate(reports, start=1): 

1045 frame = _read_onestop_part(root, kind, regime, part, variant) 

1046 (word_frames if kind == "ia" else fix_frames).append(frame) 

1047 progress.report(index, len(reports), unit="reports") 

1048 words = pd.concat(word_frames, ignore_index=True, sort=False) 

1049 fixations = pd.concat(fix_frames, ignore_index=True, sort=False) 

1050 if len(part_list) > 1: 

1051 words, fixations = _onestop_drop_unmatched_screens(words, fixations) 

1052 return _onestop_parts_as_screens(words, fixations) 

1053 

1054 

1055def _onestop_drop_unmatched_screens( 

1056 words: pd.DataFrame, fixations: pd.DataFrame 

1057) -> tuple[pd.DataFrame, pd.DataFrame]: 

1058 """Keep only the screens both reports have, loudly (DATA-63). 

1059 

1060 Several parts make each part a screen, and ``multipart.validate_matching_parts`` 

1061 rejects a screen present in one report and absent from the other — which 

1062 the OSF release has: its *Answers* fixation report holds 31 fixations of 

1063 ``l55_519``'s first reading of ``2_9_2_Ele`` that its interest-area report 

1064 has no row for. One such gap would otherwise abort the whole load, as 

1065 ``_multipleye_drop_screens_without_boxes`` already guards against for 

1066 MultiplEYE. A frame without the identity columns is returned untouched. 

1067 """ 

1068 keys = ["participant_id", "unique_trial_id", "part"] 

1069 if not (set(keys) <= set(words.columns) and set(keys) <= set(fixations.columns)): 

1070 return words, fixations 

1071 word_keys = pd.MultiIndex.from_frame(words[keys].astype(str)) 

1072 fix_keys = pd.MultiIndex.from_frame(fixations[keys].astype(str)) 

1073 shared = word_keys.unique().intersection(fix_keys.unique()) 

1074 keep_words = word_keys.isin(shared) 

1075 keep_fix = fix_keys.isin(shared) 

1076 for name, frame, keep in ( 

1077 ("word box", words, keep_words), 

1078 ("fixation", fixations, keep_fix), 

1079 ): 

1080 if not keep.all(): 

1081 dropped = frame.loc[~keep, keys].drop_duplicates() 

1082 _LOGGER.warning( 

1083 "OneStop: dropped %d %s row(s) on %d screen(s) the other report " 

1084 "does not have (e.g. %s).", 

1085 int((~keep).sum()), 

1086 name, 

1087 len(dropped), 

1088 ", ".join( 

1089 "/".join(map(str, row)) for row in dropped.head(3).to_numpy() 

1090 ), 

1091 ) 

1092 # A fresh index: later steps align on it, and a gapped one is how a 

1093 # positional assignment quietly lands on the wrong rows. 

1094 return ( 

1095 words.loc[keep_words].reset_index(drop=True), 

1096 fixations.loc[keep_fix].reset_index(drop=True), 

1097 ) 

1098 

1099 

1100def load_onestop( 

1101 root, 

1102 *, 

1103 regime: str = "ordinary", 

1104 parts: Iterable[str] | None = None, 

1105 variant: str = "public", 

1106 download: bool = False, 

1107 names: str = "source", 

1108) -> tuple[pd.DataFrame, pd.DataFrame]: 

1109 """Load OneStop as normalized ``(words, fixations)`` frames, ready to plot. 

1110 

1111 Under OneStop's own column names; ``names="canonical"`` for the internal 

1112 ones (see `api.load_scanpath_data`). 

1113 

1114 ``root`` is a folder holding (or to download into, public variant only) the 

1115 OneStop reports. Narrow the load with ``regime`` (``ordinary`` / 

1116 ``information_seeking`` / ``repeated`` / ``information_seeking_repeated``), 

1117 ``parts`` (any subset of ``Title / Question_Preview / Paragraph / Questions / 

1118 Answers / QA / Feedback`` — default Paragraph), and ``variant`` (``public`` 

1119 OSF release or ``lacclab`` local export). The public OSF reports are large; 

1120 pass ``download=True`` to fetch the chosen regime + parts into ``root`` on 

1121 first use:: 

1122 

1123 import scanpath_studio as sps 

1124 

1125 words, fixations = sps.load_onestop( 

1126 "data/OneStop", regime="ordinary", parts=["Paragraph"], download=True 

1127 ) 

1128 pid, tid = sps.list_trials(words, fixations).iloc[0] # one reading 

1129 fig = sps.plot_scanpath(words, fixations, pid, tid, canvas_size=(2560, 1440)) 

1130 

1131 OneStop's presentation monitor was 2560×1440 px (Dell U2715H; Berzak et al. 

1132 2025, Methods → Apparatus); pass that as ``canvas_size`` to 

1133 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] for true-to-scale rendering. 

1134 The reports already match the bundled demo's schema, so this reuses the generic 

1135 auto-detect → normalize path (no OneStop-specific column mapping). 

1136 """ 

1137 words_raw, fixations_raw = onestop_raw_frames( 

1138 root, regime=regime, parts=parts, variant=variant, download=download 

1139 ) 

1140 

1141 from . import api 

1142 

1143 return api.load_scanpath_data(words=words_raw, fixations=fixations_raw, names=names) 

1144 

1145 

1146# --------------------------------------------------------------------------- 

1147# MultiplEYE — multilingual eye-tracking-while-reading corpus 

1148# (https://multipleye.eu). Tested against the read-only ZH/Chinese Zurich 

1149# sample under ``data/MultiplEYE_ZH_CH_Zurich_1_2025``. 

1150# --------------------------------------------------------------------------- 

1151# 

1152# Why a dedicated loader (the generic Upload flow can't do this on its own): 

1153# 

1154# * **Identity is in the path, not the columns.** Per-session folders 

1155# ``{pid}_ZH_CH_1_ET{1|2}/`` (under ``fixations/`` and ``scanpaths/``) hold 

1156# one comma-CSV per trial; none of the files carry a participant / trial / 

1157# stimulus column. We parse all four from the folder + file name. 

1158# * **ET1 and ET2 read disjoint stimuli**, so a *reader* is the whole session 

1159# string — ``participant_id = "001_ZH_CH_1_ET1"`` — not the bare pid. 

1160# * **One stimulus spans several screens** ``page_1..page_N`` (plus the 

1161# comprehension-question screens that follow them) which all reuse the *same* 

1162# on-screen coordinates. DATA-24 models that directly: one reading of one 

1163# stimulus is **one trial** (``trial_id == text_id == "<stimulus>"``) whose 

1164# screens are the corpus's own ``page`` values — ``screen_id = "page_1"`` / 

1165# ``"question_4111"`` — each its own coordinate space (``multipart.py``). 

1166# ``screen_index`` is ranked from **that reader's own fixation onsets**, never 

1167# from the screen name: reading pages are shown in page order but the question 

1168# order is *shuffled per reader*. ``screen_kind`` (``reading`` | ``question``) 

1169# labels the two. ``familiarity_rating_screen_*`` / ``subject_difficulty_screen`` 

1170# stay out: the corpus ships no AOI file for them, and 

1171# ``multipart.validate_matching_parts`` rejects word-box-less screens by design. 

1172# * **Word boxes ship once per stimulus** (no participant) as *character*-level 

1173# AOI files ``stimuli_*/aoi_stimuli_*/<stimulus>_aoi.csv``. We aggregate chars 

1174# to one bounding box per (page, word_idx). ``word_idx`` is unique within a 

1175# page — hence within a screen — so it is the word id directly. The 

1176# **question** screens' boxes come from ``<stimulus>_aoi_questions.csv``, whose 

1177# rows are per *answer-layout version*: which version a reader saw is looked up 

1178# in ``stimuli_*/config/stimulus_order_versions_*.csv`` by their bare pid, so 

1179# question boxes are always per reader. 

1180# 

1181# Fixation source: ``scanpaths/`` (preferred — already page/word-tagged) or 

1182# ``fixations/`` (raw onset/duration/x/y/page, no word linkage). The 

1183# ``scanpaths/`` export is pre-filtered to reading pages, so **question screens 

1184# are always read from ``fixations/``** whatever ``fixation_source`` says; the 

1185# resulting mixed provenance inside one trial is visible as ``screen_kind``. 

1186 

1187MULTIPLEYE_FIXATION_SOURCES = ("scanpaths", "fixations") 

1188 

1189 

1190# Presentation monitor (px) for the ZH-CH-Zurich sample, from its lab config 

1191# (``Monitor_resolution_in_px`` / ``RESOLUTION``) — the physical screen the data 

1192# was recorded on; pass as ``canvas_size`` to :func:`scanpath_studio.plot_scanpath` 

1193# for true-to-scale rendering. 

1194MULTIPLEYE_MONITOR = (1920, 1080) 

1195 

1196# The stimulus image is smaller than the screen (config ``IMAGE_WIDTH/HEIGHT_PX``) 

1197# and was shown **centered**. The raw AOI/fixation coords are image-relative 

1198# (text starts at the image's 81/88 px margins), so the loader shifts them by the 

1199# centering offset below → they land where the participant actually saw them on 

1200# the full monitor, and the page-image background sits at that same origin. 

1201MULTIPLEYE_IMAGE_SIZE = (1310, 991) 

1202_MULTIPLEYE_IMAGE_ORIGIN = ( 

1203 (MULTIPLEYE_MONITOR[0] - MULTIPLEYE_IMAGE_SIZE[0]) / 2, # 305.0 

1204 (MULTIPLEYE_MONITOR[1] - MULTIPLEYE_IMAGE_SIZE[1]) / 2, # 44.5 

1205) 

1206 

1207 

1208def _multipleye_screen_kind(page) -> str: 

1209 """``"reading"`` / ``"question"`` for a corpus ``page`` value, else ``""``. 

1210 

1211 The empty string is what excludes a screen from the load — the rating and 

1212 difficulty screens ship no AOI file, so they would be word-box-less screens 

1213 and ``multipart.validate_matching_parts`` rejects those by design.""" 

1214 text = str(page) 

1215 if text.startswith("page_"): 

1216 return "reading" 

1217 if text.startswith("question_"): 

1218 return "question" 

1219 return "" 

1220 

1221 

1222def _multipleye_page_number(page) -> int | None: 

1223 """The 1-based page number of a ``page_N`` screen, or None.""" 

1224 text = str(page) 

1225 return int(text[5:]) if text.startswith("page_") and text[5:].isdigit() else None 

1226 

1227 

1228# The question id inside a screen name. It appears **unpadded** on the fixations 

1229# (``question_4111``) and **zero-padded to the stimulus id's width** in the AOI 

1230# file (``Lit_Alchemist_4_question_04111_target``), so the two only ever join on 

1231# ``int(question_id)`` — a string match silently finds nothing. 

1232_MULTIPLEYE_QUESTION_RE = re.compile(r"question_(?P<qid>\d+)(?:_(?P<block>.+))?$") 

1233 

1234 

1235def _multipleye_question_parts(page) -> tuple[int, str] | None: 

1236 """``(question_id, aoi_block)`` for a question screen / AOI page, or None. 

1237 

1238 ``question_4111`` → ``(4111, "stem")`` (the fixations' name and the AOI's 

1239 stem block); ``Lit_Alchemist_4_question_04111_target`` → ``(4111, "target")``. 

1240 """ 

1241 match = _MULTIPLEYE_QUESTION_RE.search(str(page)) 

1242 if match is None: 

1243 return None 

1244 return int(match.group("qid")), (match.group("block") or "stem") 

1245 

1246 

1247def _multipleye_question_screen_id(question_id: int) -> str: 

1248 """The ``screen_id`` for a question screen — the fixations' own unpadded name.""" 

1249 return f"question_{question_id}" 

1250 

1251 

1252def _multipleye_bare_pid(participant_id) -> int | None: 

1253 """The integer pid inside a session string (``001_ZH_CH_1_ET1`` → 1).""" 

1254 head = str(participant_id).split("_", 1)[0] 

1255 return int(head) if head.isdigit() else None 

1256 

1257 

1258# Per-trial file name: ``<session>_[PRACTICE_]trial_<n>_<stimulus>_<kind>`` 

1259# e.g. ``001_ZH_CH_1_ET1_trial_1_Lit_Alchemist_4_fixation`` or 

1260# ``001_ZH_CH_1_ET1_PRACTICE_trial_1_Enc_WikiMoon_13_scanpath``. 

1261_MULTIPLEYE_TRIAL_RE = re.compile( 

1262 r"^(?P<session>\d+_[A-Za-z]{2}_[A-Za-z]{2}_\d+_ET\d+)_" 

1263 r"(?:(?P<practice>PRACTICE)_)?trial_(?P<trial_num>\d+)_" 

1264 r"(?P<stimulus>.+)_(?P<kind>fixation|scanpath)$" 

1265) 

1266 

1267 

1268def _multipleye_parse_filename(stem: str) -> dict | None: 

1269 """Parse a per-trial file stem into its identity parts, or ``None``.""" 

1270 match = _MULTIPLEYE_TRIAL_RE.match(stem) 

1271 return match.groupdict() if match else None 

1272 

1273 

1274def _multipleye_aoi_dir(root: Path) -> Path: 

1275 """Locate the per-stimulus character-AOI directory under ``root``.""" 

1276 for aoi_dir in sorted(root.glob("stimuli_*/aoi_stimuli_*")): 

1277 if aoi_dir.is_dir(): 

1278 return aoi_dir 

1279 raise FileNotFoundError( 

1280 f"No MultiplEYE AOI directory found under {root} — expected " 

1281 "stimuli_*/aoi_stimuli_*/<stimulus>_aoi.csv files." 

1282 ) 

1283 

1284 

1285# The stimulus images were rendered by the MultiplEYE pipeline with the FONT_SIZE 

1286# (monitor px) + FONT declared in the stimulus config (``stimuli_*/config/ 

1287# config_*.py``). Carrying these onto the data lets the app reproduce the exact 

1288# reading text true-to-scale instead of guessing the size from box geometry and 

1289# rendering CJK in a generic fallback font. Known font files → a CSS family stack 

1290# (the actual installed family name + sensible fallbacks); unknown ones get a 

1291# humanised name + a monospace (and CJK, when the name says so) fallback. 

1292_MULTIPLEYE_FONT_CSS = { 

1293 "notosansmonocjksc": "'Noto Sans Mono CJK SC', 'Noto Sans CJK SC', monospace", 

1294 "notosansmonocjktc": "'Noto Sans Mono CJK TC', 'Noto Sans CJK TC', monospace", 

1295 "notosansmonocjkjp": "'Noto Sans Mono CJK JP', 'Noto Sans CJK JP', monospace", 

1296 "notosansmonocjkkr": "'Noto Sans Mono CJK KR', 'Noto Sans CJK KR', monospace", 

1297 "notosansmono": "'Noto Sans Mono', monospace", 

1298} 

1299_FONT_NAME_DROP = ("vf", "variable", "regular", "bold", "italic", "medium") 

1300_FONT_SIZE_RE = re.compile(r"^\s*FONT_SIZE\s*=\s*([0-9]+(?:\.[0-9]+)?)", re.MULTILINE) 

1301_FONT_FILE_RE = re.compile(r"^\s*FONT\s*=\s*[\"']([^\"']+)[\"']", re.MULTILINE) 

1302 

1303 

1304def _multipleye_config_path(root: Path) -> Path | None: 

1305 """The stimulus-generation config (``stimuli_*/config/config_*.py``), or None.""" 

1306 return next(iter(sorted(root.glob("stimuli_*/config/config_*.py"))), None) 

1307 

1308 

1309def _multipleye_font_css(font_file: str) -> str: 

1310 """CSS font-family stack for a config ``FONT`` path (e.g. a ``.ttf`` filename).""" 

1311 stem = re.sub(r"[^a-z0-9]", "", Path(font_file).stem.lower()) 

1312 for drop in _FONT_NAME_DROP: 

1313 stem = stem.removesuffix(drop) 

1314 if stem in _MULTIPLEYE_FONT_CSS: 

1315 return _MULTIPLEYE_FONT_CSS[stem] 

1316 # Unknown font: humanise the file stem (spaces at case/digit boundaries) and 

1317 # append a monospace fallback, plus a CJK fallback when the name implies CJK. 

1318 raw = Path(font_file).stem 

1319 human = re.sub(r"(?<=[a-z])(?=[A-Z])|(?<=[A-Za-z])(?=[0-9])", " ", raw).strip() 

1320 tail = "'Noto Sans CJK SC', monospace" if "cjk" in stem else "monospace" 

1321 return f"'{human}', {tail}" if human else tail 

1322 

1323 

1324def _multipleye_font_config(root: Path) -> tuple[float | None, str | None]: 

1325 """``(font_px, css_font_family)`` from the stimulus config, or ``(None, None)``. 

1326 

1327 Reads ``FONT_SIZE`` (monitor px the images were rendered at) and ``FONT`` (the 

1328 typeface) from ``config_*.py`` by regex — the file imports lab-specific paths, 

1329 so we never exec it. Returns ``(None, None)`` when no config / no FONT_SIZE. 

1330 """ 

1331 path = _multipleye_config_path(root) 

1332 if path is None: 

1333 return None, None 

1334 try: 

1335 text = path.read_text(encoding="utf-8", errors="replace") 

1336 except OSError: 

1337 return None, None 

1338 size_m = _FONT_SIZE_RE.search(text) 

1339 if size_m is None: 

1340 return None, None 

1341 font_px = float(size_m.group(1)) 

1342 file_m = _FONT_FILE_RE.search(text) 

1343 family = _multipleye_font_css(file_m.group(1)) if file_m else None 

1344 return font_px, family 

1345 

1346 

1347def _multipleye_char_boxes(chars: pd.DataFrame, group: list) -> pd.DataFrame: 

1348 """Aggregate character AOI rows to one bounding box per ``group``. 

1349 

1350 Emits *edge* columns (``left/right/top/bottom``) already shifted onto the 

1351 centered stimulus' on-screen position, so they are true-to-scale on 

1352 ``MULTIPLEYE_MONITOR``. Groups keep first-appearance order (the caller sorts 

1353 the characters into reading order first).""" 

1354 chars = chars.copy() 

1355 chars["_right"] = chars["top_left_x"] + chars["width"] 

1356 chars["_bottom"] = chars["top_left_y"] + chars["height"] 

1357 boxes = ( 

1358 chars.groupby(group, sort=False) 

1359 .agg( 

1360 left=("top_left_x", "min"), 

1361 top=("top_left_y", "min"), 

1362 right=("_right", "max"), 

1363 bottom=("_bottom", "max"), 

1364 line_idx=("line_idx", "min"), 

1365 word=("word", "first"), 

1366 ) 

1367 .reset_index() 

1368 ) 

1369 off_x, off_y = _MULTIPLEYE_IMAGE_ORIGIN 

1370 boxes["left"] += off_x 

1371 boxes["right"] += off_x 

1372 boxes["top"] += off_y 

1373 boxes["bottom"] += off_y 

1374 return boxes 

1375 

1376 

1377def _multipleye_stamp_box_identity(boxes: pd.DataFrame, stimulus: str) -> pd.DataFrame: 

1378 """Stamp the stimulus-level identity columns shared by every word box.""" 

1379 boxes["stimulus"] = stimulus 

1380 # One reading of one stimulus is one trial; the screens inside it are the 

1381 # corpus's own `page` values (mapped to `screen_id` by the schemas below). 

1382 boxes["trial_id"] = stimulus 

1383 boxes["text_id"] = stimulus 

1384 boxes["genre"] = stimulus.split("_")[0] 

1385 return boxes 

1386 

1387 

1388def _multipleye_word_boxes_from_frame( 

1389 chars: pd.DataFrame, stimulus: str 

1390) -> pd.DataFrame: 

1391 """Aggregate one stimulus' character-level AOI rows to one box per (page, word). 

1392 

1393 Reading pages only (``page_*``) — question screens come from the sibling 

1394 ``_aoi_questions`` file (:func:`_multipleye_question_boxes_from_frame`), whose 

1395 layout is reader-specific. ``stimulus`` is the (CamelCase-canonical) name 

1396 stamped into ``stimulus`` / ``text_id`` / ``genre`` / ``trial_id``. Emits 

1397 *edge* columns (``left/right/top/bottom``) to match ``MULTIPLEYE_WORD_SCHEMA`` 

1398 (no participant — stimulus-level boxes broadcast across readers). Shared by 

1399 the directory loader and the upload recipe; returns an empty frame if no 

1400 reading-page rows (or no ``page`` column at all — e.g. a stray upload).""" 

1401 if "page" not in chars.columns: 

1402 return chars.iloc[0:0] 

1403 chars = chars[chars["page"].astype(str).str.startswith("page_")].copy() 

1404 if chars.empty: 

1405 return chars 

1406 boxes = _multipleye_char_boxes(chars, ["page", "word_idx"]) 

1407 boxes["screen_kind"] = "reading" 

1408 return _multipleye_stamp_box_identity(boxes, stimulus) 

1409 

1410 

1411def _multipleye_word_boxes(aoi_dir: Path, stimuli: Iterable[str]) -> pd.DataFrame: 

1412 """Stimulus-level reading-page word boxes from the AOI files under ``aoi_dir``. 

1413 

1414 One row per (stimulus, page, word_idx); raises if a stimulus' AOI file is 

1415 missing (the directory loader knows exactly which file each stimulus needs).""" 

1416 frames = [] 

1417 for stimulus in sorted(set(stimuli)): 

1418 path = aoi_dir / f"{stimulus.lower()}_aoi.csv" 

1419 if not path.is_file(): 

1420 raise FileNotFoundError( 

1421 f"MultiplEYE AOI file not found for stimulus {stimulus!r}: {path}" 

1422 ) 

1423 frames.append(_multipleye_word_boxes_from_frame(pd.read_csv(path), stimulus)) 

1424 return pd.concat(frames, ignore_index=True) 

1425 

1426 

1427# --- Question screens: layout version → AOI blocks → per-screen word boxes --- 

1428 

1429 

1430def _multipleye_versions_path(root: Path) -> Path | None: 

1431 """The answer-layout version table under ``root``, or None.""" 

1432 return next( 

1433 iter(sorted(root.glob("stimuli_*/config/stimulus_order_versions_*.csv"))), 

1434 None, 

1435 ) 

1436 

1437 

1438def _multipleye_layout_versions(frame: pd.DataFrame | None) -> dict: 

1439 """``{bare pid -> question-image layout version}`` from a versions table. 

1440 

1441 ``stimulus_order_versions_*.csv`` has one row per ``version_number`` with an 

1442 optional ``participant_id``; only the assigned rows matter. The file keys on 

1443 the *participant*, not the session, so ET1 and ET2 share one version.""" 

1444 if frame is None or getattr(frame, "empty", True): 

1445 return {} 

1446 if not {"version_number", "participant_id"} <= set(frame.columns): 

1447 return {} 

1448 pid = pd.to_numeric(frame["participant_id"], errors="coerce") 

1449 version = pd.to_numeric(frame["version_number"], errors="coerce") 

1450 keep = pid.notna() & version.notna() 

1451 return {int(p): int(v) for p, v in zip(pid[keep], version[keep])} 

1452 

1453 

1454def _multipleye_read_layout_versions(root: Path) -> dict: 

1455 """Read the versions table under ``root`` (empty map when absent/unreadable).""" 

1456 path = _multipleye_versions_path(root) 

1457 if path is None: 

1458 return {} 

1459 try: 

1460 frame = pd.read_csv(path) 

1461 except ( 

1462 OSError, 

1463 UnicodeDecodeError, 

1464 pd.errors.ParserError, 

1465 pd.errors.EmptyDataError, 

1466 ): 

1467 return {} 

1468 return _multipleye_layout_versions(frame) 

1469 

1470 

1471_QUESTION_KEY = ["_question_id", "_aoi_block"] 

1472 

1473 

1474def _multipleye_question_block_order(chars: pd.DataFrame) -> pd.DataFrame: 

1475 """``_block_order`` per (question, AOI block), by first-character geometry. 

1476 

1477 The five blocks of a question screen (``stem`` + ``target`` + three 

1478 distractors) are laid out around the screen, so reading order is the order of 

1479 their first character: top, then left.""" 

1480 firsts = chars.groupby(_QUESTION_KEY, sort=False).head(1) 

1481 firsts = firsts.sort_values( 

1482 ["_question_id", "top_left_y", "top_left_x"], kind="stable" 

1483 ) 

1484 order = firsts[_QUESTION_KEY].copy() 

1485 order["_block_order"] = order.groupby("_question_id", sort=False).cumcount() 

1486 return order 

1487 

1488 

1489def _multipleye_question_boxes_from_frame( 

1490 chars: pd.DataFrame, stimulus: str, version: int | None 

1491) -> pd.DataFrame: 

1492 """Question-screen word boxes for one stimulus at one answer-layout version. 

1493 

1494 ``chars`` is a ``<stimulus>_aoi_questions.csv`` frame, whose ``page`` values 

1495 name the AOI block (``Lit_Alchemist_4_question_04111_target``) or the question 

1496 stem (``question_04111``) with the stimulus id **zero-padded** — the join back 

1497 to the fixations' ``question_4111`` is therefore on ``int(question_id)``. 

1498 

1499 Every block restarts ``word_idx`` at 0, so ``word_idx`` alone cannot be the 

1500 word id: the blocks are put in reading order and one counter runs across 

1501 them, with ``line_idx`` densely re-ranked the same way, so both are unique 

1502 within the screen. The block name survives as ``aoi_block``, a genuinely 

1503 useful per-word facet. Returns an empty frame when the file / version yields 

1504 nothing.""" 

1505 if chars is None or getattr(chars, "empty", True) or "page" not in chars.columns: 

1506 return pd.DataFrame() 

1507 if "question_image_version" in chars.columns: 

1508 if version is None: 

1509 return pd.DataFrame() 

1510 wanted = f"question_images_version_{int(version)}" 

1511 # Filter BEFORE copying: the file holds every layout version (250 in the 

1512 # ZH-CH sample), so this keeps well under 1% of the rows. 

1513 chars = chars[chars["question_image_version"].astype(str) == wanted] 

1514 parsed = chars["page"].map(_multipleye_question_parts) 

1515 chars = chars[parsed.notna()].copy() 

1516 if chars.empty: 

1517 return pd.DataFrame() 

1518 parsed = parsed[parsed.notna()] 

1519 chars["_question_id"] = [int(value[0]) for value in parsed] 

1520 chars["_aoi_block"] = [str(value[1]) for value in parsed] 

1521 

1522 order = _multipleye_question_block_order(chars) 

1523 sort_by = [ 

1524 column 

1525 for column in ("line_idx", "char_idx_in_line", "char_idx") 

1526 if column in chars.columns 

1527 ] 

1528 if sort_by: 

1529 chars = chars.sort_values(_QUESTION_KEY + sort_by, kind="stable") 

1530 # One aggregation for the whole stimulus/version rather than one per block: 

1531 # the groups are ~10 rows each, so per-call pandas overhead dominated. 

1532 out = _multipleye_char_boxes(chars, _QUESTION_KEY + ["word_idx"]) 

1533 out = out.merge(order, on=_QUESTION_KEY, how="left").sort_values( 

1534 ["_question_id", "_block_order", "line_idx", "word_idx"], kind="stable" 

1535 ) 

1536 per_question = out.groupby("_question_id", sort=False) 

1537 out["word_idx"] = per_question.cumcount() 

1538 # Dense per-screen line ids: the group codes run in the sorted order above, so 

1539 # subtracting each question's first code rebases them to 0. 

1540 codes = out.groupby(_QUESTION_KEY[:1] + ["_block_order", "line_idx"], sort=False) 

1541 line_key = codes.ngroup() 

1542 out["line_idx"] = line_key - line_key.groupby(out["_question_id"]).transform("min") 

1543 out["page"] = [_multipleye_question_screen_id(q) for q in out["_question_id"]] 

1544 out = out.rename(columns={"_question_id": "question_id", "_aoi_block": "aoi_block"}) 

1545 out["screen_kind"] = "question" 

1546 out = out.drop(columns=["_block_order"]).reset_index(drop=True) 

1547 return _multipleye_stamp_box_identity(out, stimulus) 

1548 

1549 

1550def _stamp_multipleye_fixations( 

1551 df: pd.DataFrame, info: dict, *, kinds: tuple[str, ...] = ("reading",) 

1552) -> pd.DataFrame: 

1553 """Filter to the wanted screen kinds and stamp identity parsed from a filename. 

1554 

1555 ``info`` is a ``_multipleye_parse_filename`` dict; ``kinds`` selects which 

1556 screens survive (``reading`` and/or ``question`` — see 

1557 :func:`_multipleye_screen_kind`; every other screen is always dropped). Rows 

1558 keep the corpus's own ``page`` as the ``screen_id`` and gain ``screen_kind``; 

1559 ``trial_id`` is the stimulus, since one reading of a stimulus is one trial. 

1560 ``name == 'fixation'`` filtering applies to the ``scanpaths`` export, which 

1561 tags each row. Returns an empty frame if nothing remains (or the upload has no 

1562 ``page`` column at all). Shared by the directory loader (one file) and the 

1563 upload recipe (one source_file group).""" 

1564 if "page" not in df.columns: 

1565 return df.iloc[0:0] 

1566 screen_kind = df["page"].map(_multipleye_screen_kind) 

1567 df = df[screen_kind.isin(kinds)] 

1568 if "name" in df.columns: # scanpaths tag each row; keep fixations only 

1569 df = df[df["name"] == "fixation"] 

1570 if df.empty: 

1571 return df 

1572 df = df.copy() 

1573 df["screen_kind"] = screen_kind.loc[df.index] 

1574 session = info["session"] 

1575 stimulus = info["stimulus"] 

1576 # Shift image-relative fixation coords to the centered on-screen position. 

1577 off_x, off_y = _MULTIPLEYE_IMAGE_ORIGIN 

1578 df["location_x"] = pd.to_numeric(df["location_x"], errors="coerce") + off_x 

1579 df["location_y"] = pd.to_numeric(df["location_y"], errors="coerce") + off_y 

1580 df["participant_id"] = session 

1581 df["participant"] = session.split("_", 1)[0] # bare pid 

1582 df["session"] = session.rsplit("_", 1)[-1] # ET1 / ET2 

1583 df["stimulus"] = stimulus 

1584 df["text_id"] = stimulus # stimulus-level grouping key (see word boxes) 

1585 df["genre"] = stimulus.split("_")[0] 

1586 df["is_practice"] = bool(info["practice"]) 

1587 df["trial_num"] = int(info["trial_num"]) 

1588 df["trial_id"] = stimulus 

1589 # The presentation monitor, so `multipart.screen_canvas_size` reports the real 

1590 # screen for every screen instead of inferring one from the data's extent. 

1591 df["canvas_width"] = MULTIPLEYE_MONITOR[0] 

1592 df["canvas_height"] = MULTIPLEYE_MONITOR[1] 

1593 return df 

1594 

1595 

1596def _multipleye_fixations( 

1597 root: Path, 

1598 source: str, 

1599 sessions: Iterable[str] | None, 

1600 stimuli: Iterable[str] | None, 

1601 *, 

1602 include_question_screens: bool = True, 

1603) -> pd.DataFrame: 

1604 """Concatenated per-trial fixations, tagged with parsed identity columns. 

1605 

1606 Identity (participant / session / stimulus → ``trial_id``, plus the screen) 

1607 comes from the folder + file name. Reading pages come from ``source`` 

1608 (the word-tagged ``scanpaths/`` by preference; ``fixations/`` works too, with 

1609 no ``word_idx``). **Question screens always come from ``fixations/``** — the 

1610 ``scanpaths/`` export is pre-filtered to reading pages — so a default load 

1611 keeps its word-tagged reading fixations and still shows the question screens; 

1612 ``screen_kind`` marks the resulting mixed provenance.""" 

1613 base = root / source 

1614 suffix = "scanpath" if source == "scanpaths" else "fixation" 

1615 session_filter = None if sessions is None else {str(s) for s in sessions} 

1616 stim_filter = None if stimuli is None else {str(s) for s in stimuli} 

1617 raw_base = root / "fixations" 

1618 want_questions = include_question_screens and ( 

1619 source == "fixations" or raw_base.is_dir() 

1620 ) 

1621 kinds = ( 

1622 ("reading", "question") 

1623 if source == "fixations" and want_questions 

1624 else ("reading",) 

1625 ) 

1626 

1627 def _wanted(path: Path) -> dict | None: 

1628 info = _multipleye_parse_filename(path.stem) 

1629 if info is None: 

1630 return None 

1631 if stim_filter is not None and info["stimulus"] not in stim_filter: 

1632 return None 

1633 return info 

1634 

1635 # UX-166 fix-round-1 (Minor #6): keep only files `_wanted` will actually 

1636 # read, so "N of M files" isn't inflated by ones a session/stimulus filter 

1637 # (or an unparseable name) was always going to skip. 

1638 reading = [ 

1639 (path, info) 

1640 for session_dir in sorted(p for p in base.iterdir() if p.is_dir()) 

1641 if session_filter is None or session_dir.name in session_filter 

1642 for path in sorted(session_dir.glob(f"*_{suffix}.csv")) 

1643 if (info := _wanted(path)) is not None 

1644 ] 

1645 # UX-166: "0 of N" before the first file, so a gated card is armed from the 

1646 # start — a report only after each file would hide the first one. 

1647 progress.report(0, len(reading), unit="files") 

1648 frames = [] 

1649 for index, (path, info) in enumerate(reading, start=1): 

1650 stamped = _stamp_multipleye_fixations(pd.read_csv(path), info, kinds=kinds) 

1651 if not stamped.empty: 

1652 frames.append(stamped) 

1653 progress.report(index, len(reading), unit="files") 

1654 if not frames: 

1655 raise FileNotFoundError( 

1656 f"No MultiplEYE {source} files matched under {base} " 

1657 f"(sessions={sessions}, stimuli={stimuli})." 

1658 ) 

1659 if want_questions and source != "fixations": 

1660 for session_dir in sorted(p for p in raw_base.iterdir() if p.is_dir()): 

1661 if session_filter is not None and session_dir.name not in session_filter: 

1662 continue 

1663 for path in sorted(session_dir.glob("*_fixation.csv")): 

1664 info = _wanted(path) 

1665 if info is None: 

1666 continue 

1667 stamped = _stamp_multipleye_fixations( 

1668 pd.read_csv(path), info, kinds=("question",) 

1669 ) 

1670 if not stamped.empty: 

1671 frames.append(stamped) 

1672 return pd.concat(frames, ignore_index=True, sort=False) 

1673 

1674 

1675# --- Screen ordering: the reader's own onsets, never the screen name ---------- 

1676 

1677 

1678def _multipleye_screen_order( 

1679 fixations: pd.DataFrame, *, by_onset: bool 

1680) -> pd.DataFrame: 

1681 """One row per ``(participant_id, trial_id, page)`` with its 1-based order. 

1682 

1683 ``by_onset`` ranks the screens by that reader's **first fixation onset**, 

1684 which is the only correct order: reading pages are shown in page order, but 

1685 the *question order is shuffled per reader* (``001_ZH_CH_1_ET1`` saw 

1686 ``question_4132`` before ``question_4131``). The page-number fallback exists 

1687 for the reading-pages-only load, where the word boxes stay stimulus-level and 

1688 so must carry the same, reader-independent index the fixations do. 

1689 

1690 Also returns ``_screen_onset`` — each screen's first onset, which re-zeroes 

1691 ``onset`` into the per-screen clock (``onset`` itself stays parent-global).""" 

1692 keys = ["participant_id", "trial_id", "page"] 

1693 onsets = ( 

1694 fixations.assign(_onset=pd.to_numeric(fixations["onset"], errors="coerce")) 

1695 .groupby(keys, sort=False)["_onset"] 

1696 .min() 

1697 .reset_index() 

1698 .rename(columns={"_onset": "_screen_onset"}) 

1699 ) 

1700 numbers = onsets["page"].map(_multipleye_page_number) 

1701 if not by_onset and numbers.notna().all(): 

1702 onsets["screen_index"] = numbers.astype(int) 

1703 else: 

1704 onsets["screen_index"] = ( 

1705 onsets.groupby(["participant_id", "trial_id"])["_screen_onset"] 

1706 .rank(method="first") 

1707 .astype(int) 

1708 ) 

1709 return onsets 

1710 

1711 

1712# Explicit schemas from the raw MultiplEYE frames to the canonical schema (no 

1713# participant on words — stimulus-level boxes broadcast across readers). The 

1714# corpus's own ``page`` is the ``screen_id``, so one trial keeps every screen of 

1715# a reading in its own coordinate space (``data._copy_screen_fields`` already 

1716# accepts all of these keys, so nothing in data.py needs plumbing for them). 

1717MULTIPLEYE_WORD_SCHEMA = dict( 

1718 participant=None, 

1719 trial="trial_id", 

1720 text_id="text_id", 

1721 word_id="word_idx", 

1722 text="word", 

1723 line="line_idx", 

1724 left="left", 

1725 right="right", 

1726 top="top", 

1727 bottom="bottom", 

1728 screen_id="page", 

1729 screen_index="screen_index", 

1730) 

1731MULTIPLEYE_FIX_SCHEMA = dict( 

1732 participant="participant_id", 

1733 trial="trial_id", 

1734 text_id="text_id", 

1735 duration="duration", 

1736 timestamp="onset", 

1737 x="location_x", 

1738 y="location_y", 

1739 word_id="word_idx", 

1740 screen_id="page", 

1741 screen_index="screen_index", 

1742 # `onset` stays the parent-global clock; this is it re-zeroed per screen. 

1743 screen_timestamp="screen_timestamp_ms", 

1744 canvas_width="canvas_width", 

1745 canvas_height="canvas_height", 

1746) 

1747# When per-reader reading measures or question screens are present, the word 

1748# boxes carry a real ``participant_id`` (per reader) and the per-reader IA_* 

1749# measures, so they take the participant branch in ``normalize_words`` (no 

1750# stimulus-level broadcast). 

1751MULTIPLEYE_WORD_SCHEMA_PER_READER = dict( 

1752 MULTIPLEYE_WORD_SCHEMA, participant="participant_id" 

1753) 

1754 

1755 

1756def multipleye_word_schema(words: pd.DataFrame) -> dict: 

1757 """The word schema matching a raw MultiplEYE word frame's shape. 

1758 

1759 Per-reader boxes (reading measures and/or question screens attached) carry a 

1760 ``participant_id`` and must NOT be broadcast; stimulus-level boxes must.""" 

1761 return dict( 

1762 MULTIPLEYE_WORD_SCHEMA_PER_READER 

1763 if "participant_id" in getattr(words, "columns", ()) 

1764 else MULTIPLEYE_WORD_SCHEMA 

1765 ) 

1766 

1767 

1768# --- Side data: questions / reader metadata / reading measures / page images --- 

1769 

1770# Reader-metadata columns carried from participant_data.csv → namespaced ``pp_*``. 

1771MULTIPLEYE_PARTICIPANT_META_COLS = { 

1772 "age": "pp_age", 

1773 "gender": "pp_gender", 

1774 "native_language_1": "pp_native_language", 

1775 "years_education": "pp_years_education", 

1776 "level_education": "pp_education_level", 

1777} 

1778 

1779# MultiplEYE reading-measure column → the canonical EyeLink IA_* name the app 

1780# already recognizes (``data.WORD_OPTIONAL_FIELDS``) and prefers over recomputed 

1781# measures. Regression in/out *flags* are derived from the counts (RR is 

1782# "re-reading", not a regression flag, so it is intentionally not mapped). 

1783MULTIPLEYE_RM_MAP = { 

1784 "FFD": "IA_FIRST_FIXATION_DURATION", 

1785 "FPRT": "IA_FIRST_RUN_DWELL_TIME", # first-pass / gaze duration 

1786 "TFT": "IA_DWELL_TIME", # total fixation time 

1787 "TFC": "IA_FIXATION_COUNT", 

1788 "RPD_inc": "IA_REGRESSION_PATH_DURATION", 

1789 "TRC_in": "IA_REGRESSION_IN_COUNT", 

1790 "TRC_out": "IA_REGRESSION_OUT_COUNT", 

1791 "skipped": "IA_SKIP", 

1792} 

1793 

1794# Reading-measures file name: the stimulus part has NO trailing ``_<id>``. 

1795_MULTIPLEYE_RM_RE = re.compile( 

1796 r"^(?P<session>\d+_[A-Za-z]{2}_[A-Za-z]{2}_\d+_ET\d+)_" 

1797 r"(?:PRACTICE_)?trial_\d+_(?P<stim_name>.+)_reading_measures$" 

1798) 

1799 

1800 

1801def _multipleye_questions_path(root: Path) -> Path | None: 

1802 """The comprehension-questions workbook under ``root``, or None.""" 

1803 return next( 

1804 iter(sorted(root.glob("stimuli_*/multipleye_comprehension_questions_*.xlsx"))), 

1805 None, 

1806 ) 

1807 

1808 

1809def _multipleye_image_dir(root: Path) -> tuple[Path, str] | None: 

1810 """``(stimulus-images dir, language tag)`` or None. 

1811 

1812 The language is read from the directory name (``stimuli_images_zh_ch_1`` → 

1813 ``zh``), never hardcoded, so other MultiplEYE languages work.""" 

1814 for d in sorted(root.glob("stimuli_*/stimuli_images_*")): 

1815 if d.is_dir(): 

1816 parts = d.name.removeprefix("stimuli_images_").split("_") 

1817 return d, (parts[0] if parts and parts[0] else "") 

1818 return None 

1819 

1820 

1821def _multipleye_question_image_dir(root: Path) -> tuple[Path, str] | None: 

1822 """``(question-images dir, language tag)`` or None. 

1823 

1824 Holds one ``question_images_version_<N>/`` per answer layout; the images are 

1825 1310x991 — the same size as a page image — so the underlay origin is 

1826 ``_MULTIPLEYE_IMAGE_ORIGIN`` for question screens too.""" 

1827 for d in sorted(root.glob("stimuli_*/question_images_*")): 

1828 if d.is_dir(): 

1829 parts = d.name.removeprefix("question_images_").split("_") 

1830 return d, (parts[0] if parts and parts[0] else "") 

1831 return None 

1832 

1833 

1834def _multipleye_questions_from_frame(qs: pd.DataFrame) -> dict: 

1835 """``{stimulus -> JSON list of comprehension questions}`` from a questions 

1836 frame (workbook sheet 0), joined by ``stimulus_name + "_" + stimulus_id``.""" 

1837 if ( 

1838 qs is None 

1839 or qs.empty 

1840 or not {"stimulus_name", "stimulus_id"} <= set(qs.columns) 

1841 ): 

1842 return {} 

1843 sort_cols = [c for c in ("condition_no", "question_no") if c in qs.columns] 

1844 out: dict = {} 

1845 for (name, sid), group in qs.groupby(["stimulus_name", "stimulus_id"]): 

1846 if sort_cols: 

1847 group = group.sort_values(sort_cols) 

1848 items = [] 

1849 for _, r in group.iterrows(): 

1850 distractors = [ 

1851 str(r[c]).strip() 

1852 for c in ("distractor_a", "distractor_b", "distractor_c") 

1853 if c in qs.columns 

1854 and str(r.get(c, "nan")).strip().lower() not in ("", "nan") 

1855 ] 

1856 items.append( 

1857 { 

1858 "question": str(r.get("question", "")), 

1859 "target": str(r.get("target", "")), 

1860 "distractors": distractors, 

1861 "condition": str(r.get("condition_name", "")), 

1862 "question_no": ( 

1863 int(r["question_no"]) 

1864 if "question_no" in qs.columns 

1865 and pd.notna(r.get("question_no")) 

1866 else None 

1867 ), 

1868 } 

1869 ) 

1870 out[f"{name}_{int(sid)}"] = json.dumps(items, ensure_ascii=False) 

1871 return out 

1872 

1873 

1874def _multipleye_questions_by_stimulus(xlsx_path: Path) -> dict: 

1875 """Read the comprehension workbook and return ``{stimulus -> questions JSON}``. 

1876 

1877 Comprehension questions are optional enrichment (a per-stimulus JSON column); 

1878 reading the ``.xlsx`` needs the optional ``openpyxl`` dependency. If it isn't 

1879 installed, skip the enrichment (empty map) rather than failing the whole load 

1880 — the scanpaths/word-boxes don't depend on it.""" 

1881 try: 

1882 frame = pd.read_excel(xlsx_path, sheet_name=0) 

1883 except ImportError: 

1884 return {} 

1885 return _multipleye_questions_from_frame(frame) 

1886 

1887 

1888def _normalize_multipleye_participant_meta( 

1889 df: pd.DataFrame | None, 

1890) -> pd.DataFrame | None: 

1891 """Select + namespace reader-metadata columns from a participant_data frame. 

1892 

1893 One row per ``(participant_id:Int64, session:str)``; None if the join keys 

1894 are absent. Used by both the directory loader and the upload path.""" 

1895 if df is None or not {"participant_id", "session"} <= set(df.columns): 

1896 return None 

1897 keep = { 

1898 src: dest 

1899 for src, dest in MULTIPLEYE_PARTICIPANT_META_COLS.items() 

1900 if src in df.columns 

1901 } 

1902 out = df[["participant_id", "session", *keep]].copy() 

1903 out["participant_id"] = pd.to_numeric( 

1904 out["participant_id"], errors="coerce" 

1905 ).astype("Int64") 

1906 out["session"] = out["session"].astype(str) 

1907 return out.rename(columns=keep).drop_duplicates(["participant_id", "session"]) 

1908 

1909 

1910def _multipleye_participant_meta(root: Path) -> pd.DataFrame | None: 

1911 """Reader metadata from ``participant_data.csv`` (namespaced ``pp_*``), or None.""" 

1912 path = root / "participant_data.csv" 

1913 return ( 

1914 _normalize_multipleye_participant_meta(pd.read_csv(path)) 

1915 if path.is_file() 

1916 else None 

1917 ) 

1918 

1919 

1920def _merge_multipleye_participant_meta( 

1921 fixations: pd.DataFrame, meta: pd.DataFrame | None 

1922) -> pd.DataFrame: 

1923 """Left-merge reader metadata onto fixations by ``(int(participant), session)``. 

1924 

1925 The bare pid is zero-padded text on the fixations (``001``) but an integer in 

1926 participant_data (``1``) — join on the int-coerced value, never the string.""" 

1927 if meta is None or fixations.empty: 

1928 return fixations 

1929 fixations = fixations.copy() 

1930 fixations["_pid_int"] = pd.to_numeric( 

1931 fixations["participant"], errors="coerce" 

1932 ).astype("Int64") 

1933 meta = meta.rename(columns={"participant_id": "_pid_int"}) 

1934 merged = fixations.merge(meta, on=["_pid_int", "session"], how="left") 

1935 return merged.drop(columns=["_pid_int"]) 

1936 

1937 

1938def _multipleye_read_reading_measures( 

1939 root: Path, 

1940 sessions: Iterable[str] | None, 

1941 stim_namemap: dict, 

1942) -> pd.DataFrame: 

1943 """Per-(reader, page, word) reading measures, columns renamed to IA_*. 

1944 

1945 ``stim_namemap`` resolves the id-stripped file-name stimulus (``Lit_Alchemist``) 

1946 to the full stimulus (``Lit_Alchemist_4``). Empty frame if none found.""" 

1947 base = root / "reading_measures" 

1948 session_filter = None if sessions is None else {str(s) for s in sessions} 

1949 keep_src = list(MULTIPLEYE_RM_MAP) 

1950 frames = [] 

1951 for session_dir in sorted(p for p in base.iterdir() if p.is_dir()): 

1952 if session_filter is not None and session_dir.name not in session_filter: 

1953 continue 

1954 for path in sorted(session_dir.glob("*_reading_measures.csv")): 

1955 m = _MULTIPLEYE_RM_RE.match(path.stem) 

1956 if m is None: 

1957 continue 

1958 stimulus = stim_namemap.get(m.group("stim_name")) 

1959 if stimulus is None: # a stimulus not in this load (stimuli filter) 

1960 continue 

1961 df = pd.read_csv(path) 

1962 if "page" not in df.columns or "word_idx" not in df.columns: 

1963 continue 

1964 cols = ["page", "word_idx"] + [c for c in keep_src if c in df.columns] 

1965 df = df[cols].rename(columns=MULTIPLEYE_RM_MAP) 

1966 if "IA_REGRESSION_IN_COUNT" in df.columns: 

1967 df["IA_REGRESSION_IN"] = (df["IA_REGRESSION_IN_COUNT"] > 0).astype(int) 

1968 if "IA_REGRESSION_OUT_COUNT" in df.columns: 

1969 df["IA_REGRESSION_OUT"] = (df["IA_REGRESSION_OUT_COUNT"] > 0).astype( 

1970 int 

1971 ) 

1972 df["participant_id"] = m.group("session") 

1973 df["stimulus"] = stimulus 

1974 frames.append(df) 

1975 return ( 

1976 pd.concat(frames, ignore_index=True, sort=False) if frames else pd.DataFrame() 

1977 ) 

1978 

1979 

1980def _multipleye_words_per_reader( 

1981 stim_boxes: pd.DataFrame, 

1982 rm: pd.DataFrame, 

1983 fixations: pd.DataFrame, 

1984 question_boxes: pd.DataFrame | None = None, 

1985) -> pd.DataFrame: 

1986 """Reading-page boxes per reader, scoped to the screens they actually fixated. 

1987 

1988 The stimulus-level boxes are replicated onto the ``(reader, stimulus, page)`` 

1989 triples present in that reader's fixations — an **inner** join, so a page the 

1990 reader skipped never becomes an orphan screen for 

1991 ``multipart.validate_matching_parts`` to reject. That reader's pre-aggregated 

1992 reading measures merge on ``(participant, stimulus, page, word_idx)`` 

1993 (``word_idx`` restarts per page, so page MUST be in the key). Question-screen 

1994 boxes are already per reader and are appended as they are.""" 

1995 if stim_boxes.empty: 

1996 words = stim_boxes 

1997 else: 

1998 pairs = fixations[["participant_id", "stimulus", "page"]].drop_duplicates() 

1999 words = pairs.merge(stim_boxes, on=["stimulus", "page"], how="inner") 

2000 if not rm.empty: 

2001 words = words.merge( 

2002 rm, on=["participant_id", "stimulus", "page", "word_idx"], how="left" 

2003 ) 

2004 if question_boxes is not None and not question_boxes.empty: 

2005 words = pd.concat([words, question_boxes], ignore_index=True, sort=False) 

2006 return words 

2007 

2008 

2009def _multipleye_question_word_boxes( 

2010 aoi_source, fixations: pd.DataFrame, versions: dict 

2011) -> pd.DataFrame: 

2012 """Per-(reader, stimulus) question-screen word boxes. 

2013 

2014 ``aoi_source`` is either the corpus AOI directory (files are read on demand) 

2015 or a ``{stimulus -> question-AOI frame}`` map (the upload path). Which answer 

2016 layout a reader saw is looked up in ``versions`` by their **bare pid**, so a 

2017 reader with no assigned version — or a stimulus with no question-AOI rows — 

2018 contributes no boxes, and the caller drops those question screens rather than 

2019 guessing a layout that would draw plausible boxes in the wrong places.""" 

2020 if fixations.empty or "screen_kind" not in fixations.columns: 

2021 return pd.DataFrame() 

2022 wanted = fixations[fixations["screen_kind"] == "question"] 

2023 if wanted.empty: 

2024 return pd.DataFrame() 

2025 if not versions: 

2026 _LOGGER.warning( 

2027 "MultiplEYE: no answer-layout version table, so the %d question " 

2028 "screen(s) are skipped (their AOI layout is reader-specific).", 

2029 wanted[["participant_id", "trial_id", "page"]].drop_duplicates().shape[0], 

2030 ) 

2031 return pd.DataFrame() 

2032 

2033 # Two memos, because they have different granularities: the corpus assigns a 

2034 # DISTINCT layout version to every participant, so a per-(stimulus, version) 

2035 # box cache almost never hits — but the file it parses is per stimulus and is 

2036 # the expensive part (~15 MB each), so that read is memoized separately. 

2037 files: dict = {} 

2038 cache: dict = {} 

2039 frames = [] 

2040 for (reader, stimulus), group in wanted.groupby( 

2041 ["participant_id", "stimulus"], sort=False 

2042 ): 

2043 stimulus = str(stimulus) 

2044 version = versions.get(_multipleye_bare_pid(reader)) 

2045 key = (stimulus, version) 

2046 if key not in cache: 

2047 if version is None: 

2048 # No assigned layout: don't even read the file for it. 

2049 cache[key] = pd.DataFrame() 

2050 else: 

2051 if stimulus not in files: 

2052 files[stimulus] = _multipleye_question_aoi_frame( 

2053 aoi_source, stimulus 

2054 ) 

2055 cache[key] = _multipleye_question_boxes_from_frame( 

2056 files[stimulus], stimulus, version 

2057 ) 

2058 boxes = cache[key] 

2059 if boxes.empty: 

2060 _LOGGER.warning( 

2061 "MultiplEYE: no question AOI rows for reader %s on %s " 

2062 "(layout version %s) — its question screens are skipped.", 

2063 reader, 

2064 stimulus, 

2065 version, 

2066 ) 

2067 continue 

2068 seen = set(group["page"].astype(str)) 

2069 boxes = boxes[boxes["page"].isin(seen)].copy() 

2070 boxes["participant_id"] = reader 

2071 frames.append(boxes) 

2072 return ( 

2073 pd.concat(frames, ignore_index=True, sort=False) if frames else pd.DataFrame() 

2074 ) 

2075 

2076 

2077def _multipleye_question_aoi_frame(aoi_source, stimulus: str) -> pd.DataFrame | None: 

2078 """One stimulus' question-AOI rows, from a directory or an uploaded map.""" 

2079 if isinstance(aoi_source, dict): 

2080 return aoi_source.get(stimulus.lower()) 

2081 path = Path(aoi_source) / f"{stimulus.lower()}_aoi_questions.csv" 

2082 return pd.read_csv(path) if path.is_file() else None 

2083 

2084 

2085def _multipleye_drop_screens_without_boxes( 

2086 words: pd.DataFrame, fixations: pd.DataFrame 

2087) -> pd.DataFrame: 

2088 """Drop fixations on screens that have no word boxes, loudly. 

2089 

2090 ``multipart.validate_matching_parts`` rejects a screen present in one report 

2091 and absent from the other, so a screen we could not build boxes for (a 

2092 question screen whose layout version is unknown, a stimulus whose AOI file 

2093 was not uploaded) must be dropped here rather than crashing the whole load 

2094 inside ``data.harmonize_frames``.""" 

2095 if words.empty or fixations.empty or "page" not in words.columns: 

2096 return fixations 

2097 keys = ["trial_id", "page"] 

2098 if "participant_id" in words.columns: 

2099 keys.insert(0, "participant_id") 

2100 present = words[keys].drop_duplicates().astype(str) 

2101 present["_has_boxes"] = True 

2102 probe = fixations[keys].astype(str) 

2103 merged = probe.merge(present, on=keys, how="left") 

2104 keep = merged["_has_boxes"].notna().to_numpy(dtype=bool) 

2105 if not keep.all(): 

2106 dropped = fixations.loc[~keep, ["trial_id", "page"]].drop_duplicates() 

2107 _LOGGER.warning( 

2108 "MultiplEYE: dropped %d fixation(s) on %d screen(s) with no word " 

2109 "boxes (e.g. %s).", 

2110 int((~keep).sum()), 

2111 len(dropped), 

2112 ", ".join(f"{t}/{p}" for t, p in dropped.head(3).to_numpy()), 

2113 ) 

2114 return fixations.loc[keep] 

2115 

2116 

2117def _multipleye_apply_screen_order( 

2118 words: pd.DataFrame, fixations: pd.DataFrame, *, by_onset: bool 

2119) -> tuple[pd.DataFrame, pd.DataFrame]: 

2120 """Stamp ``screen_index`` (+ the per-screen clock) on both frames. 

2121 

2122 Ranks only the **included** screens, so the indices stay a contiguous 1..N 

2123 once the rating / difficulty screens are gone. Per-reader word boxes take the 

2124 reader's own ranking; stimulus-level boxes (reading pages only, no 

2125 participant column to key on) take the page number, which is the same order 

2126 for every reader and — crucially — the same value the fixations get, so 

2127 ``multipart.part_catalog`` sees no conflict between the two reports.""" 

2128 order = _multipleye_screen_order(fixations, by_onset=by_onset) 

2129 keys = ["participant_id", "trial_id", "page"] 

2130 fixations = fixations.merge(order, on=keys, how="left") 

2131 fixations["screen_timestamp_ms"] = ( 

2132 pd.to_numeric(fixations["onset"], errors="coerce") - fixations["_screen_onset"] 

2133 ) 

2134 fixations = fixations.drop(columns=["_screen_onset"]) 

2135 if words.empty or "page" not in words.columns: 

2136 return words, fixations 

2137 if "participant_id" in words.columns: 

2138 words = words.merge(order.drop(columns=["_screen_onset"]), on=keys, how="left") 

2139 else: 

2140 numbers = words["page"].map(_multipleye_page_number) 

2141 if numbers.notna().all(): 

2142 words = words.assign(screen_index=numbers.astype(int)) 

2143 return words, fixations 

2144 

2145 

2146def _multipleye_stamp_questions(df: pd.DataFrame, questions: dict) -> pd.DataFrame: 

2147 """Stamp the per-stimulus comprehension-questions JSON onto a frame.""" 

2148 if df.empty or not questions or "stimulus" not in df.columns: 

2149 return df 

2150 df = df.copy() 

2151 df["comprehension_questions"] = df["stimulus"].map(questions) 

2152 return df 

2153 

2154 

2155def _multipleye_stamp_image_path( 

2156 df: pd.DataFrame, image_dir: Path, lang: str 

2157) -> pd.DataFrame: 

2158 """Stamp the per-(stimulus, reading page) stimulus-image path onto a frame. 

2159 

2160 ``Lit_Alchemist_4`` + ``page_3`` → ``…/lit_alchemist_id4_page_3_<lang>.png``. 

2161 Question screens are left blank here — their image lives in a per-reader 

2162 layout directory (:func:`_multipleye_stamp_question_image_path`).""" 

2163 if df.empty or not {"stimulus", "page"} <= set(df.columns): 

2164 return df 

2165 df = df.copy() 

2166 stim = df["stimulus"].astype(str) 

2167 name = stim.str.rsplit("_", n=1).str[0].str.lower() 

2168 sid = stim.str.rsplit("_", n=1).str[1] 

2169 page = df["page"].astype(str) 

2170 pnum = page.str.replace("page_", "", regex=False) 

2171 df["image_path"] = ( 

2172 f"{image_dir}/" + name + "_id" + sid + "_page_" + pnum + f"_{lang}.png" 

2173 ).where(page.str.startswith("page_")) 

2174 # Where the (centered) image sits on the monitor — matches the coordinate 

2175 # offset applied to the fixations/boxes, so the image aligns with the data. 

2176 df["image_x"] = _MULTIPLEYE_IMAGE_ORIGIN[0] 

2177 df["image_y"] = _MULTIPLEYE_IMAGE_ORIGIN[1] 

2178 return df 

2179 

2180 

2181def _multipleye_stamp_question_image_path( 

2182 df: pd.DataFrame, image_dir: Path, lang: str, versions: dict 

2183) -> pd.DataFrame: 

2184 """Fill ``image_path`` for question screens from the reader's layout version. 

2185 

2186 ``Lit_Alchemist_4`` + ``question_4111`` for a reader on version 71 → 

2187 ``…/question_images_version_71/Lit_Alchemist_id4_question_04111_<lang>.png``. 

2188 The stimulus name keeps its CamelCase here (unlike the lowercase page images) 

2189 and the question id is zero-padded to five digits, as the corpus writes it.""" 

2190 needed = {"stimulus", "page", "participant_id"} 

2191 if df.empty or not needed <= set(df.columns) or not versions: 

2192 return df 

2193 df = df.copy() 

2194 keys = ["participant_id", "stimulus", "page"] 

2195 triples = df.loc[ 

2196 df["page"].astype(str).str.startswith("question_"), keys 

2197 ].drop_duplicates() 

2198 if triples.empty: 

2199 return df 

2200 resolved = [] 

2201 for reader, stimulus, page in triples.astype(str).to_numpy(): 

2202 version = versions.get(_multipleye_bare_pid(reader)) 

2203 parts = _multipleye_question_parts(page) 

2204 if version is None or parts is None: 

2205 continue 

2206 name, _, sid = str(stimulus).rpartition("_") 

2207 resolved.append( 

2208 { 

2209 "participant_id": reader, 

2210 "stimulus": stimulus, 

2211 "page": page, 

2212 "_question_image": ( 

2213 f"{image_dir}/question_images_version_{int(version)}/" 

2214 f"{name}_id{sid}_question_{parts[0]:05d}_{lang}.png" 

2215 ), 

2216 } 

2217 ) 

2218 if "image_path" not in df.columns: 

2219 df["image_path"] = pd.NA 

2220 if resolved: 

2221 lookup = pd.DataFrame(resolved) 

2222 probe = df[keys].astype(str).merge(lookup, on=keys, how="left") 

2223 found = probe["_question_image"].notna().to_numpy() 

2224 df.loc[found, "image_path"] = probe.loc[found, "_question_image"].to_numpy() 

2225 df["image_x"] = _MULTIPLEYE_IMAGE_ORIGIN[0] 

2226 df["image_y"] = _MULTIPLEYE_IMAGE_ORIGIN[1] 

2227 return df 

2228 

2229 

2230def _multipleye_stamp_font( 

2231 df: pd.DataFrame, font_px: float | None, family: str | None 

2232) -> pd.DataFrame: 

2233 """Stamp the stimulus typeface (``stimulus_font_px`` / ``stimulus_font_family``). 

2234 

2235 The values are dataset-constant (read once from the stimulus config); the app 

2236 snaps its font controls to them so the reading text renders at the exact size 

2237 and typeface the stimulus images were drawn with.""" 

2238 if df.empty or font_px is None: 

2239 return df 

2240 df = df.copy() 

2241 df["stimulus_font_px"] = float(font_px) 

2242 if family: 

2243 df["stimulus_font_family"] = family 

2244 return df 

2245 

2246 

2247def multipleye_inventory( 

2248 root, *, fixation_source: str = "scanpaths" 

2249) -> tuple[tuple[str, ...], tuple[str, ...]]: 

2250 """(sessions, stimuli) available under a MultiplEYE ``root``. 

2251 

2252 Cheap directory scan (filenames only, no CSV reads) for the app's 

2253 session/stimulus pickers. Returns sorted tuples; empties if ``root`` has no 

2254 recognizable per-trial files.""" 

2255 root = Path(root) 

2256 base = root / fixation_source 

2257 if not base.is_dir(): 

2258 base = root / next( 

2259 (s for s in MULTIPLEYE_FIXATION_SOURCES if (root / s).is_dir()), "" 

2260 ) 

2261 sessions: set = set() 

2262 stimuli: set = set() 

2263 if base.is_dir(): 

2264 for session_dir in (p for p in base.iterdir() if p.is_dir()): 

2265 for path in session_dir.glob("*.csv"): 

2266 info = _multipleye_parse_filename(path.stem) 

2267 if info is None: 

2268 continue 

2269 sessions.add(info["session"]) 

2270 stimuli.add(info["stimulus"]) 

2271 return tuple(sorted(sessions)), tuple(sorted(stimuli)) 

2272 

2273 

2274def multipleye_raw_frames( 

2275 root, 

2276 *, 

2277 sessions: Iterable[str] | None = None, 

2278 stimuli: Iterable[str] | None = None, 

2279 fixation_source: str = "scanpaths", 

2280 attach_reading_measures: bool = True, 

2281 include_question_screens: bool = True, 

2282) -> tuple[pd.DataFrame, pd.DataFrame]: 

2283 """Raw (pre-normalization) MultiplEYE ``(words, fixations)`` frames. 

2284 

2285 Same inputs as :func:`load_multipleye`, but returns the frames *before* 

2286 schema normalization — for callers that run their own auto-detection / 

2287 column mapping (e.g. the Streamlit app's MultiplEYE data source). Fixations 

2288 carry parsed ``participant_id`` (the session), ``trial_id`` (the stimulus), 

2289 the screen (``page`` + ``screen_index`` + ``screen_kind``), pixel 

2290 ``location_x/y``, the trial-level facets (``genre`` / ``session`` / 

2291 ``is_practice`` / ``trial_num``), and any reader metadata 

2292 (``pp_*`` from participant_data.csv), comprehension questions, and stimulus 

2293 image path that the corpus ships. 

2294 

2295 Word boxes are stimulus-level (no participant → broadcast) *unless* 

2296 ``reading_measures/`` exists and ``attach_reading_measures`` is on, or 

2297 question screens are included — in which case the boxes are emitted **per 

2298 reader**, with the corpus's pre-aggregated reading measures merged in as 

2299 ``IA_*`` columns (the app then prefers them over recomputed metrics) and the 

2300 reader's own question-screen boxes appended. ``include_question_screens`` 

2301 (on by default) opts out of the comprehension-question screens; they are 

2302 also skipped when the answer-layout version table is missing, since guessing 

2303 a layout would draw entirely plausible boxes in the wrong places. Use 

2304 :func:`load_multipleye` for normalized frames. 

2305 """ 

2306 root = Path(root) 

2307 if fixation_source not in MULTIPLEYE_FIXATION_SOURCES: 

2308 raise ValueError( 

2309 f"fixation_source must be one of {list(MULTIPLEYE_FIXATION_SOURCES)}, " 

2310 f"got {fixation_source!r}" 

2311 ) 

2312 if not (root / fixation_source).is_dir(): 

2313 # Fall back to whichever per-trial source folder exists. 

2314 alt = next( 

2315 (s for s in MULTIPLEYE_FIXATION_SOURCES if (root / s).is_dir()), None 

2316 ) 

2317 if alt is None: 

2318 raise FileNotFoundError( 

2319 f"No MultiplEYE fixation data under {root} — expected a " 

2320 f"'scanpaths' or 'fixations' folder of per-session subfolders." 

2321 ) 

2322 fixation_source = alt 

2323 

2324 fixations = _multipleye_fixations( 

2325 root, 

2326 fixation_source, 

2327 sessions, 

2328 stimuli, 

2329 include_question_screens=include_question_screens, 

2330 ) 

2331 # Reader metadata (age/gender/languages…) merged onto every fixation row. 

2332 fixations = _merge_multipleye_participant_meta( 

2333 fixations, _multipleye_participant_meta(root) 

2334 ) 

2335 

2336 aoi_dir = _multipleye_aoi_dir(root) 

2337 stim_boxes = _multipleye_word_boxes(aoi_dir, fixations["stimulus"].unique()) 

2338 versions = _multipleye_read_layout_versions(root) 

2339 question_boxes = ( 

2340 _multipleye_question_word_boxes(aoi_dir, fixations, versions) 

2341 if include_question_screens 

2342 else pd.DataFrame() 

2343 ) 

2344 # Pre-aggregated reading measures → per-reader word boxes (skips the 

2345 # stimulus-level broadcast). Only when the corpus ships reading_measures/. 

2346 # Question screens force the same shape: their layout is reader-specific. 

2347 per_reader = attach_reading_measures and (root / "reading_measures").is_dir() 

2348 if per_reader or not question_boxes.empty: 

2349 rm = pd.DataFrame() 

2350 if per_reader: 

2351 namemap = {s.rsplit("_", 1)[0]: s for s in fixations["stimulus"].unique()} 

2352 rm = _multipleye_read_reading_measures(root, sessions, namemap) 

2353 words = _multipleye_words_per_reader(stim_boxes, rm, fixations, question_boxes) 

2354 else: 

2355 words = stim_boxes 

2356 

2357 # Screens we could not build boxes for would be orphans in harmonize_frames, 

2358 # so drop them (loudly) before the screen order is ranked — that keeps the 

2359 # index a contiguous 1..N over exactly the screens that survive. 

2360 fixations = _multipleye_drop_screens_without_boxes(words, fixations) 

2361 words, fixations = _multipleye_apply_screen_order( 

2362 words, fixations, by_onset="participant_id" in words.columns 

2363 ) 

2364 

2365 # Comprehension questions + stimulus images, stamped on both frames. 

2366 qpath = _multipleye_questions_path(root) 

2367 if qpath is not None: 

2368 qmap = _multipleye_questions_by_stimulus(qpath) 

2369 words = _multipleye_stamp_questions(words, qmap) 

2370 fixations = _multipleye_stamp_questions(fixations, qmap) 

2371 image = _multipleye_image_dir(root) 

2372 if image is not None: 

2373 image_dir, lang = image 

2374 words = _multipleye_stamp_image_path(words, image_dir, lang) 

2375 fixations = _multipleye_stamp_image_path(fixations, image_dir, lang) 

2376 question_image = _multipleye_question_image_dir(root) 

2377 if question_image is not None: 

2378 image_dir, lang = question_image 

2379 words = _multipleye_stamp_question_image_path(words, image_dir, lang, versions) 

2380 fixations = _multipleye_stamp_question_image_path( 

2381 fixations, image_dir, lang, versions 

2382 ) 

2383 # Reading typeface (size + family) from the stimulus config → the app renders 

2384 # the text true-to-scale at the exact font the images were drawn with. 

2385 font_px, font_family = _multipleye_font_config(root) 

2386 words = _multipleye_stamp_font(words, font_px, font_family) 

2387 fixations = _multipleye_stamp_font(fixations, font_px, font_family) 

2388 return words, fixations 

2389 

2390 

2391def load_multipleye( 

2392 root, 

2393 *, 

2394 sessions: Iterable[str] | None = None, 

2395 stimuli: Iterable[str] | None = None, 

2396 fixation_source: str = "scanpaths", 

2397 include_question_screens: bool = True, 

2398 names: str = "source", 

2399) -> tuple[pd.DataFrame, pd.DataFrame]: 

2400 """Load MultiplEYE as normalized ``(words, fixations)`` frames, ready to plot. 

2401 

2402 Under the loader's column names; ``names="canonical"`` for the internal 

2403 ones (see `api.load_scanpath_data`). 

2404 

2405 ``root`` is a MultiplEYE session set (e.g. 

2406 ``data/MultiplEYE_ZH_CH_Zurich_1_2025``). Narrow the load with ``sessions`` 

2407 (full session ids, e.g. ``["001_ZH_CH_1_ET1"]``) and/or ``stimuli`` (e.g. 

2408 ``["Lit_Alchemist_4"]``). 

2409 

2410 Participants are session ids (ET1 and ET2 read disjoint stimuli, so each is 

2411 a distinct reader). A trial is **one reading of one stimulus** 

2412 (``trial_id == text_id == "Lit_Alchemist_4"``), and its screens — the reading 

2413 pages ``page_1…page_N`` plus the comprehension-question screens — are 

2414 ``screen_id`` values in presentation order (``screen_index``, ranked from 

2415 that reader's own fixation onsets, since the question order is shuffled per 

2416 reader). ``screen_kind`` is ``reading`` or ``question``:: 

2417 

2418 words, fixations = load_multipleye( 

2419 "data/MultiplEYE_ZH_CH_Zurich_1_2025", stimuli=["Lit_Alchemist_4"] 

2420 ) 

2421 fig = scanpath_studio.plot_scanpath( 

2422 words, fixations, screen="page_1", canvas_size=MULTIPLEYE_MONITOR 

2423 ) 

2424 

2425 ``fixation_source`` is ``"scanpaths"`` (default; fixations pre-tagged with 

2426 page + word index) or ``"fixations"`` (raw, no word linkage); question 

2427 screens always come from ``fixations/``, which is the only export that keeps 

2428 them. ``include_question_screens=False`` loads the reading pages alone. 

2429 """ 

2430 words_raw, fixations_raw = multipleye_raw_frames( 

2431 root, 

2432 sessions=sessions, 

2433 stimuli=stimuli, 

2434 fixation_source=fixation_source, 

2435 include_question_screens=include_question_screens, 

2436 ) 

2437 

2438 from . import api 

2439 

2440 return api.load_scanpath_data( 

2441 words=words_raw, 

2442 fixations=fixations_raw, 

2443 word_schema=multipleye_word_schema(words_raw), 

2444 fix_schema=dict( 

2445 MULTIPLEYE_FIX_SCHEMA, 

2446 word_id="word_idx" if "word_idx" in fixations_raw.columns else None, 

2447 ), 

2448 names=names, 

2449 ) 

2450 

2451 

2452MULTIPLEYE_DATA_DIR_ENV = "MULTIPLEYE_DATA_DIR" 

2453MULTIPLEYE_BUNDLE_FIXATION_SOURCE = "scanpaths" 

2454 

2455 

2456def multipleye_bundle_dir() -> Path | None: 

2457 """Resolve the configured MultiplEYE raw-export root, if any.""" 

2458 raw = os.environ.get(MULTIPLEYE_DATA_DIR_ENV, "").strip() 

2459 if not raw: 

2460 from .constants import MULTIPLEYE_BUNDLE_DEFAULT_DIR 

2461 

2462 raw = MULTIPLEYE_BUNDLE_DEFAULT_DIR.strip() 

2463 return Path(raw) if raw else None 

2464 

2465 

2466def _resolve_multipleye_session( 

2467 root: Path, participant: str, fixation_source: str 

2468) -> str | None: 

2469 """Match a case-insensitive deep-link participant to its session id.""" 

2470 sessions, _ = multipleye_inventory(root, fixation_source=fixation_source) 

2471 wanted = participant.strip().lower() 

2472 return next((session for session in sessions if session.lower() == wanted), None) 

2473 

2474 

2475def load_multipleye_server_bundle( 

2476 participant: str | None = None, 

2477) -> tuple[pd.DataFrame, pd.DataFrame]: 

2478 """Load the configured raw MultiplEYE export for the app review source. 

2479 

2480 A participant narrows the load to its case-insensitively matched session; 

2481 omitting it loads the full export. Missing roots return empty frames, while 

2482 malformed exports and unknown named sessions raise a descriptive error for 

2483 the UI boundary to display. 

2484 """ 

2485 root = multipleye_bundle_dir() 

2486 if root is None or not root.is_dir(): 

2487 return pd.DataFrame(), pd.DataFrame() 

2488 sessions = None 

2489 if participant: 

2490 session = _resolve_multipleye_session( 

2491 root, participant, MULTIPLEYE_BUNDLE_FIXATION_SOURCE 

2492 ) 

2493 if session is None: 

2494 raise ValueError( 

2495 f"No MultiplEYE session matching participant {participant!r} " 

2496 f"under {root / MULTIPLEYE_BUNDLE_FIXATION_SOURCE}." 

2497 ) 

2498 sessions = [session] 

2499 return multipleye_raw_frames( 

2500 root, 

2501 sessions=sessions, 

2502 stimuli=None, 

2503 fixation_source=MULTIPLEYE_BUNDLE_FIXATION_SOURCE, 

2504 ) 

2505 

2506 

2507# --- Browser-upload path ----------------------------------------------------- 

2508# Loading MultiplEYE through the Add-dataset wizard: the browser strips folders, 

2509# so identity is recovered from each row's ``source_file`` (the uploaded filename 

2510# stem, tagged by ``data.read_tables``) instead of the directory tree. 

2511 

2512 

2513def _multipleye_aoi_stimulus_from_source(stem: str) -> str: 

2514 """Stimulus name from an AOI filename stem, stripping a trailing ``_aoi``. 

2515 

2516 ``lit_alchemist_4_aoi`` → ``lit_alchemist_4`` (still lowercase; the caller 

2517 canonicalizes the case). ``_aoi_questions`` files (question/option AOIs, whose 

2518 rows aren't ``page_*`` and so produce no word boxes) are handled too.""" 

2519 s = str(stem) 

2520 for suffix in ("_aoi_questions", "_aoi"): 

2521 if s.lower().endswith(suffix): 

2522 return s[: -len(suffix)] 

2523 return s 

2524 

2525 

2526def _multipleye_is_versions_upload(stem: str, group: pd.DataFrame) -> bool: 

2527 """Whether an uploaded file group is the answer-layout version table.""" 

2528 if "stimulus_order_versions" in str(stem).lower(): 

2529 return True 

2530 return "version_number" in group.columns and "page" not in group.columns 

2531 

2532 

2533def _multipleye_fixations_from_frame( 

2534 fixations_df: pd.DataFrame, *, include_question_screens: bool = True 

2535) -> pd.DataFrame: 

2536 """Identity-stamped fixations from a concatenated UPLOAD frame. 

2537 

2538 Rows must carry a ``source_file`` column (the uploaded filename stem). Each 

2539 file group is parsed with ``_multipleye_parse_filename``; groups whose name 

2540 isn't MultiplEYE-shaped are skipped. When both a ``_scanpath`` and a 

2541 ``_fixation`` file are uploaded for the same (session, trial), the **reading** 

2542 pages come from the scanpath one (it carries word indices) and the **question** 

2543 screens from the fixation one — mirroring the directory loader, whose 

2544 ``scanpaths/`` export is pre-filtered to reading pages. Returns an empty frame 

2545 if nothing matched (the wizard then surfaces a problem rather than 

2546 crashing).""" 

2547 from .data import SOURCE_FILE_COLUMN, source_file_name 

2548 

2549 if SOURCE_FILE_COLUMN not in fixations_df.columns: 

2550 return pd.DataFrame() 

2551 groups: dict = {} 

2552 for stem, group in fixations_df.groupby(SOURCE_FILE_COLUMN, sort=False): 

2553 # A shared file name is qualified in its label (`data.source_labels`). 

2554 info = _multipleye_parse_filename(source_file_name(stem)) 

2555 if info is None: 

2556 continue 

2557 key = (info["session"], info["trial_num"], info["stimulus"]) 

2558 groups.setdefault(key, {})[info["kind"]] = (info, group) 

2559 

2560 frames = [] 

2561 for by_kind in groups.values(): 

2562 reading = by_kind.get("scanpath") or by_kind.get("fixation") 

2563 # Question screens only ever survive in the raw fixation export. 

2564 question = by_kind.get("fixation") if include_question_screens else None 

2565 for source, kinds in ((reading, ("reading",)), (question, ("question",))): 

2566 if source is None: 

2567 continue 

2568 stamped = _stamp_multipleye_fixations(source[1], source[0], kinds=kinds) 

2569 if not stamped.empty: 

2570 frames.append(stamped) 

2571 return ( 

2572 pd.concat(frames, ignore_index=True, sort=False) if frames else pd.DataFrame() 

2573 ) 

2574 

2575 

2576def multipleye_frames_from_uploads( 

2577 fixations_df: pd.DataFrame, 

2578 aoi_df: pd.DataFrame | None = None, 

2579 *, 

2580 questions_df: pd.DataFrame | None = None, 

2581 participant_meta_df: pd.DataFrame | None = None, 

2582 versions_df: pd.DataFrame | None = None, 

2583 include_question_screens: bool = True, 

2584) -> tuple[pd.DataFrame, pd.DataFrame]: 

2585 """Raw MultiplEYE ``(words, fixations)`` frames from UPLOADED files. 

2586 

2587 The browser-upload analogue of :func:`multipleye_raw_frames`: identity is 

2588 parsed from each row's ``source_file`` (the uploaded filename stem) instead of 

2589 the directory tree, since browser uploads drop folders. ``fixations_df`` is 

2590 the concatenated scanpath/fixation CSVs; ``aoi_df`` the concatenated AOI CSVs 

2591 (optional — without it you get fixations and no word boxes), which may mix the 

2592 per-stimulus ``*_aoi.csv`` reading boxes, the ``*_aoi_questions.csv`` 

2593 question boxes, and ``stimulus_order_versions_*.csv`` — each is routed by its 

2594 filename. AOI filenames are lowercase (``lit_alchemist_4_aoi``) while scanpath 

2595 filenames are CamelCase, so each AOI group's stimulus is relabeled to the 

2596 CamelCase name seen in the fixations before building ``trial_id`` — otherwise 

2597 the stimulus-words broadcast (which inner-joins on ``trial_id``) drops every 

2598 box. 

2599 

2600 **Question screens need both** the ``*_aoi_questions.csv`` of their stimulus 

2601 **and** the versions table (as ``versions_df`` or inside ``aoi_df``): which 

2602 answer layout a reader saw is otherwise unknowable, and picking an arbitrary 

2603 one would draw entirely plausible boxes in the wrong places. Without it the 

2604 question screens are dropped with a warning. ``questions_df`` (the 

2605 comprehension workbook) and ``participant_meta_df`` (participant_data.csv) are 

2606 merged when provided. Reading measures + stimulus images need the directory 

2607 tree, so they are not available on this path. Feed the result through 

2608 :func:`load_multipleye_uploads`.""" 

2609 from .data import SOURCE_FILE_COLUMN, source_file_name 

2610 

2611 fixations = _multipleye_fixations_from_frame( 

2612 fixations_df, include_question_screens=include_question_screens 

2613 ) 

2614 fixations = _merge_multipleye_participant_meta( 

2615 fixations, _normalize_multipleye_participant_meta(participant_meta_df) 

2616 ) 

2617 qmap = ( 

2618 _multipleye_questions_from_frame(questions_df) 

2619 if questions_df is not None 

2620 else {} 

2621 ) 

2622 fixations = _multipleye_stamp_questions(fixations, qmap) 

2623 

2624 has_aoi = aoi_df is not None and not getattr(aoi_df, "empty", True) 

2625 if fixations.empty or not has_aoi or SOURCE_FILE_COLUMN not in aoi_df.columns: 

2626 if not fixations.empty: 

2627 fixations = _multipleye_apply_screen_order( 

2628 pd.DataFrame(), fixations, by_onset=True 

2629 )[1] 

2630 return pd.DataFrame(), fixations 

2631 

2632 # CamelCase canonical per lowercased stimulus, taken from the fixations. 

2633 casemap: dict = {} 

2634 for stim in fixations["stimulus"].unique(): 

2635 casemap.setdefault(str(stim).lower(), str(stim)) 

2636 

2637 frames = [] 

2638 question_aoi: dict = {} 

2639 versions = _multipleye_layout_versions(versions_df) 

2640 for stem, group in aoi_df.groupby(SOURCE_FILE_COLUMN, sort=False): 

2641 stem = source_file_name(stem) 

2642 if _multipleye_is_versions_upload(str(stem), group): 

2643 versions = versions or _multipleye_layout_versions(group) 

2644 continue 

2645 lower_stim = _multipleye_aoi_stimulus_from_source(str(stem)) 

2646 canonical = casemap.get(lower_stim.lower(), lower_stim) 

2647 if str(stem).lower().endswith("_aoi_questions"): 

2648 question_aoi[canonical.lower()] = group 

2649 continue 

2650 boxes = _multipleye_word_boxes_from_frame(group, canonical) 

2651 if not boxes.empty: 

2652 frames.append(boxes) 

2653 words = pd.concat(frames, ignore_index=True) if frames else pd.DataFrame() 

2654 

2655 question_boxes = ( 

2656 _multipleye_question_word_boxes(question_aoi, fixations, versions) 

2657 if include_question_screens and question_aoi 

2658 else pd.DataFrame() 

2659 ) 

2660 if not question_boxes.empty: 

2661 words = _multipleye_words_per_reader( 

2662 words, pd.DataFrame(), fixations, question_boxes 

2663 ) 

2664 words = _multipleye_stamp_questions(words, qmap) 

2665 fixations = _multipleye_drop_screens_without_boxes(words, fixations) 

2666 words, fixations = _multipleye_apply_screen_order( 

2667 words, fixations, by_onset="participant_id" in words.columns 

2668 ) 

2669 return words, fixations 

2670 

2671 

2672def load_multipleye_uploads( 

2673 fixations_df: pd.DataFrame, 

2674 aoi_df: pd.DataFrame | None = None, 

2675 *, 

2676 questions_df: pd.DataFrame | None = None, 

2677 participant_meta_df: pd.DataFrame | None = None, 

2678 versions_df: pd.DataFrame | None = None, 

2679 include_question_screens: bool = True, 

2680 names: str = "source", 

2681) -> tuple[pd.DataFrame, pd.DataFrame]: 

2682 """Normalized ``(words, fixations)`` from UPLOADED MultiplEYE files. 

2683 

2684 Like :func:`load_multipleye`, but for in-memory uploaded frames (identity from 

2685 each row's ``source_file``; see :func:`multipleye_frames_from_uploads`). 

2686 ``words`` is an empty frame when no AOI files were uploaded; the fixations then 

2687 plot at their own ``location_x/y`` with no word boxes. Optional 

2688 ``questions_df`` / ``participant_meta_df`` add the comprehension panel + reader 

2689 metadata, and ``versions_df`` (or a versions table inside ``aoi_df``) unlocks 

2690 the question screens. Raises ``ValueError`` (via 

2691 :func:`api.load_scanpath_data`) if the fixations frame has no 

2692 MultiplEYE-shaped filenames.""" 

2693 words_raw, fix_raw = multipleye_frames_from_uploads( 

2694 fixations_df, 

2695 aoi_df, 

2696 questions_df=questions_df, 

2697 participant_meta_df=participant_meta_df, 

2698 versions_df=versions_df, 

2699 include_question_screens=include_question_screens, 

2700 ) 

2701 

2702 from . import api 

2703 

2704 return api.load_scanpath_data( 

2705 words=words_raw if not words_raw.empty else None, 

2706 fixations=fix_raw if not fix_raw.empty else None, 

2707 word_schema=multipleye_word_schema(words_raw) if not words_raw.empty else None, 

2708 fix_schema=( 

2709 dict( 

2710 MULTIPLEYE_FIX_SCHEMA, 

2711 word_id="word_idx" if "word_idx" in fix_raw.columns else None, 

2712 ) 

2713 if not fix_raw.empty 

2714 else None 

2715 ), 

2716 names=names, 

2717 )