Coverage for scanpath_studio/data.py: 97%

2596 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1from __future__ import annotations 

2 

3import glob 

4import hashlib 

5import io 

6import logging 

7import os 

8import re 

9import string 

10import threading 

11import uuid 

12import warnings 

13import weakref 

14import zipfile 

15from collections import OrderedDict 

16from collections.abc import Callable, Hashable, Iterable, Sequence 

17from dataclasses import dataclass, field 

18from importlib import resources 

19from pathlib import Path 

20from typing import Any 

21 

22import numpy as np 

23import pandas as pd 

24import streamlit as st 

25 

26from . import progress 

27from .constants import ( 

28 DEFAULT_FIGURE_SIZE, 

29 PACKAGE_NAME, 

30 SAMPLE_INDEX, 

31 UPLOAD_FILE_TYPES, 

32 plural, 

33) 

34from .multipart import ( 

35 CANVAS_HEIGHT, 

36 CANVAS_WIDTH, 

37 PARENT_KEY, 

38 SCREEN_FIXATION_ID, 

39 SCREEN_ID, 

40 SCREEN_INDEX, 

41 SCREEN_TIMESTAMP, 

42 grouping_columns, 

43 normalize_screen_identity, 

44 validate_matching_parts, 

45) 

46 

47_LOGGER = logging.getLogger(__name__) 

48 

49# PERF-3. Per-run memo, `id(frame) -> (frame, fingerprint)`. 

50# 

51# Once the expensive subtabs went lazy the biggest remaining cost in a rerun was 

52# the cache keys themselves: ~26 calls, 43% of the run, and the same handful of 

53# frame OBJECTS over and over — every `_c_*` wrapper re-fingerprints the words 

54# and fixations frames `app.main` loaded once. Hashing the same object twice in 

55# one run cannot produce two answers, so the second hash onwards is pure waste. 

56# 

57# Entries hold a **weak** reference to the frame, and the `is` re-check below is 

58# what makes an `id()` key safe: a collected frame's id may be reissued to 

59# another object, but the entry's ref is dead by then, so the lookup misses. A 

60# strong reference would work too — and did, at first — but it pins every frame 

61# it has seen. That is worse than it sounds here: `st.cache_data` hands out a 

62# fresh object per call, so each run's corpus frames are new objects, and a 

63# strong memo kept up to `_FINGERPRINT_MEMO_MAX` of them alive from the end of 

64# one run until the top of the next — i.e. for a session's whole idle time, on 

65# frames that used to be freed the moment `main()` returned. 

66# 

67# `threading.local` scopes it to the ScriptRunner thread — one per Streamlit 

68# session — so two sessions never share entries and one session's reset can't 

69# drop another's. `reset_fingerprint_memo()` at the top of `app.main` bounds the 

70# staleness window to a single script run. (Widget callbacks run *before* the 

71# script body, so the first fingerprints of a rerun — `controls._compute_trial_filters` 

72# — still see the previous run's entries. Harmless: an id hit implies object 

73# identity either way, and those frames are the ones about to be reused.) 

74# 

75# THE ASSUMPTION: a fingerprinted frame is not mutated **in place** part-way 

76# through a run. That holds today — the frames the app fingerprints are built by 

77# `normalize_*` / `filter_*` / `.copy()` and then only read; helpers that add 

78# columns (`aggregation.py`) do it to a local copy. A frame with an *assigned* 

79# fingerprint (`_STABLE_FINGERPRINTS` below) relies on the same thing for as 

80# long as it lives, since its ID is never re-checked against its content. 

81# If you ever add an in-place `frame[col] = …` to a fingerprinted frame, either 

82# copy instead or the caches downstream of it will serve pre-mutation results. 

83_FINGERPRINT_MEMO = threading.local() 

84#: Backstop for a non-Streamlit caller (headless `api.py`, the CLI) that never 

85#: reaches `reset_fingerprint_memo`: keep the memo from growing without bound. 

86#: A rerun makes ~26 calls, so the app never reaches this; a bulk export does 

87#: (`compute_word_metrics` adds two entries per trial), which is why eviction is 

88#: least-recently-used rather than clear-everything — the latter threw away the 

89#: hot corpus frames to make room for per-trial temporaries. 

90_FINGERPRINT_MEMO_MAX = 64 

91 

92 

93#: PERF-10 → BUG-103: fingerprints the app *knows* rather than computes, so a 

94#: large frame is not re-hashed on every rerun. `id → (weakref, fingerprint | 

95#: None)`; the weak ref is what makes the id key safe (a reissued id finds a 

96#: dead ref), so an entry lives exactly as long as its frame. Three kinds: 

97#: 

98#: * **assigned** (`assign_fingerprint`): an ID from where the frame came from — 

99#: a loader's source token (`stamp_source` / `adopt_source`), a `frame_cache` 

100#: slot + key, or a derivation and its parents (`assign_derived`). Free, and 

101#: exact as long as the ID determines the content, which each producer 

102#: guarantees. 

103#: * **vouched** (`vouch_for_frames`, value `None`): a long-lived frame nothing 

104#: writes into (tests/test_frame_immutability.py) — hashed in full once, the 

105#: first time it is asked for, then remembered. 

106#: * anything else is hashed in full, once per run (`_FINGERPRINT_MEMO`). 

107#: 

108#: An entry is never overwritten: one object keeps one ID, so a frame a 

109#: producer hands back unchanged (an input returned as is) keeps its own. 

110#: Process-wide on purpose — every kind depends only on content. 

111_STABLE_FINGERPRINTS: dict[int, tuple[weakref.ref, tuple | None]] = {} 

112#: Dead entries are swept once the registry grows past this. Not a cap — an 

113#: entry is only bookkeeping for a frame that is still alive. 

114_STABLE_FINGERPRINTS_SWEEP = 64 

115#: UX-166 fix-round-2 (Ruling T5-6): guards every iteration/mutation of 

116#: `_STABLE_FINGERPRINTS` above. The dict is process-wide, so two script runs 

117#: can reach `_register_fingerprints` at once — a superseded run's build publishing 

118#: beside the run that replaced it, or two sessions' runs — though no build is 

119#: ever shared *across* sessions (`frame_cache`'s identity includes the 

120#: session's own store). An unlocked `.items()` iteration racing another 

121#: thread's insert raised `RuntimeError: dictionary changed size during 

122#: iteration`. Plain `.get` reads (`frame_fingerprint` below) need no lock under 

123#: the GIL — only the sweep-and-insert and the write-back do. 

124_STABLE_FINGERPRINTS_LOCK = threading.Lock() 

125 

126 

127def _frame_parts(value) -> list[tuple[object, pd.DataFrame]]: 

128 """``(label, frame)`` for each non-empty frame in a frame, dict or tuple.""" 

129 if isinstance(value, pd.DataFrame): 

130 items = [(0, value)] 

131 elif isinstance(value, dict): 

132 items = list(value.items()) 

133 elif isinstance(value, (tuple, list)): 

134 items = list(enumerate(value)) 

135 else: 

136 return [] 

137 return [ 

138 (label, frame) 

139 for label, frame in items 

140 if isinstance(frame, pd.DataFrame) and not frame.empty 

141 ] 

142 

143 

144def _register_fingerprints(entries: list[tuple[pd.DataFrame, tuple | None]]) -> None: 

145 """Record known fingerprints, never replacing one a frame already has.""" 

146 with _STABLE_FINGERPRINTS_LOCK: 

147 if len(_STABLE_FINGERPRINTS) > _STABLE_FINGERPRINTS_SWEEP: 

148 for dead in [ 

149 k for k, (ref, _) in _STABLE_FINGERPRINTS.items() if ref() is None 

150 ]: 

151 _STABLE_FINGERPRINTS.pop(dead, None) 

152 for frame, value in entries: 

153 current = _STABLE_FINGERPRINTS.get(id(frame)) 

154 if current is not None and current[0]() is frame: 

155 continue 

156 _STABLE_FINGERPRINTS[id(frame)] = (weakref.ref(frame), value) 

157 

158 

159def vouch_for_frames(value) -> None: 

160 """Mark long-lived frames as never written in place (PERF-10). 

161 

162 Each is hashed in full the first time its fingerprint is asked for, and that 

163 answer is kept for as long as the frame lives — for a frame the app holds 

164 run after run without knowing where it came from, such as a stored dataset 

165 read back from the recovery cache. 

166 """ 

167 _register_fingerprints([(frame, None) for _, frame in _frame_parts(value)]) 

168 

169 

170def assign_fingerprint(frame: pd.DataFrame, ident: Hashable) -> None: 

171 """Give ``frame`` the fingerprint ``("assigned", ident)`` instead of a hash. 

172 

173 BUG-103. ``ident`` must determine the frame's content — two frames with the 

174 same ``ident`` are taken to be identical, which is exactly what makes the 

175 caches downstream reuse a result. A frame that already has a fingerprint 

176 keeps it. 

177 """ 

178 if isinstance(frame, pd.DataFrame) and not frame.empty: 

179 _register_fingerprints([(frame, ("assigned", ident))]) 

180 

181 

182#: BUG-103: the `DataFrame.attrs` entry a cached loader labels its frames with. 

183#: Only a carrier: `st.cache_data` hands out a fresh copy per call, and `attrs` 

184#: survive the copy, so the label reaches the caller — where `adopt_source` 

185#: takes it off again before the frame goes anywhere else. Never read anywhere 

186#: but there: pandas copies `attrs` onto every frame derived from this one, so a 

187#: label left on would follow a filtered or edited copy that is not the source. 

188SOURCE_TOKEN_ATTR = "_sps_source_token" 

189 

190 

191def stamp_source(value): 

192 """Label a loader's frames with a fresh random token, for `adopt_source`. 

193 

194 Call it on the return value **inside** an ``@st.cache_data`` loader: the 

195 token is drawn once per real load, so every cache hit carries the same one 

196 and a reload — a new upload, another read plan, a cleared cache — a new one. 

197 Returns ``value``. 

198 """ 

199 token = uuid.uuid4().hex 

200 for label, frame in _frame_parts(value): 

201 frame.attrs[SOURCE_TOKEN_ATTR] = (token, label) 

202 return value 

203 

204 

205def adopt_source(*frames: pd.DataFrame | None) -> None: 

206 """Turn a loader's `stamp_source` label into each frame's fingerprint. 

207 

208 Call it where a loader's frames come out, before anything derives from them. 

209 The label comes off the frame, so nothing made from it inherits the token; 

210 the fingerprint stays with this object only. 

211 """ 

212 for frame in frames: 

213 if not isinstance(frame, pd.DataFrame): 

214 continue 

215 token = frame.attrs.pop(SOURCE_TOKEN_ATTR, None) 

216 if token is not None: 

217 assign_fingerprint(frame, ("source", token)) 

218 

219 

220def assign_derived(outputs, op: str, parents, params=None) -> None: 

221 """Fingerprint ``outputs`` by how they were made, not by hashing them. 

222 

223 For a pure step — ``outputs`` determined by ``op``, the ``parents`` frames' 

224 content and ``params`` — the ID is ``(op, parent fingerprints, params)``, 

225 labelled by each output's position. An output that is one of its parents 

226 (a step with nothing to do) keeps the parent's fingerprint. ``params`` must 

227 be hashable after `hashable_key`; when it is not, the outputs are simply 

228 hashed like any other frame. 

229 """ 

230 parent_ids = {id(p) for p in _as_frames(parents)} 

231 if all(id(frame) in parent_ids for _, frame in _frame_parts(outputs)): 

232 return # nothing new to name (a step with nothing to do) 

233 key = hashable_key(params) 

234 if not _plain(key): 

235 return 

236 parent_keys = tuple(frame_fingerprint(p) for p in _as_frames(parents)) 

237 # Digested: every cache downstream hashes its key on every call, and the 

238 # settings can be long (the filter's full trial list), so the ID stays 

239 # small however much went into it. `repr` is stable for what `hashable_key` 

240 # leaves: tuples of plain values, sets and dicts already sorted. 

241 digest = hashlib.blake2b( 

242 repr((op, parent_keys, key)).encode(), digest_size=16 

243 ).hexdigest() 

244 for label, frame in _frame_parts(outputs): 

245 if id(frame) not in parent_ids: 

246 assign_fingerprint(frame, ("derived", op, digest, label)) 

247 

248 

249#: What a derived ID's settings may hold: values whose `repr` is their content. 

250#: Anything else — an object whose `repr` is its address, a Series — could 

251#: name two different settings the same way, so it is hashed instead. 

252_PLAIN_TYPES = (str, int, float, bool, type(None), np.generic, pd.Timestamp) 

253 

254 

255def _plain(value) -> bool: 

256 if isinstance(value, tuple): 

257 return all(_plain(v) for v in value) 

258 return isinstance(value, _PLAIN_TYPES) 

259 

260 

261def _as_frames(value) -> tuple: 

262 if isinstance(value, pd.DataFrame) or value is None: 

263 return (value,) 

264 return tuple(value) 

265 

266 

267def hashable_key(value): 

268 """``value`` as a hashable, order-free key part (sets and dicts sorted).""" 

269 if isinstance(value, dict): 

270 return tuple(sorted(((str(k), hashable_key(v)) for k, v in value.items()))) 

271 if isinstance(value, (set, frozenset)): 

272 return tuple(sorted((hashable_key(v) for v in value), key=repr)) 

273 if isinstance(value, (list, tuple)): 

274 return tuple(hashable_key(v) for v in value) 

275 return value 

276 

277 

278#: Session-state home of the no-copy frame caches (PERF-6), one entry per slot. 

279_FRAME_CACHE_KEY = "_sps_frame_cache" 

280 

281#: UX-166 "latest request wins" (T5-1): the store key `(_LATEST_REQUESTED, 

282#: slot)` — a tuple, so it can never collide with a real (string) slot name — 

283#: holds the most recently *requested* key for that slot, recorded by 

284#: `frame_cache` on every call, hit or miss. A build that finishes only writes 

285#: `store[slot]` while this still names its own key; otherwise a newer request 

286#: has already been *made* — whether or not it has itself finished yet, or 

287#: ever will — and this build's (still-valid, still returned to its own 

288#: caller) result must not clobber it. 

289_LATEST_REQUESTED = "__requested__" 

290 

291#: PERF-18: the store key `(_EARLIER, slot)` holds a slot's *earlier* entries 

292#: — `(key, value)` pairs, most recent first — when it was asked to `keep` 

293#: more than one. `store[slot]` stays the current entry, so a one-entry slot 

294#: looks exactly as it always did. 

295_EARLIER = "__earlier__" 

296 

297 

298@dataclass 

299class _InFlight: 

300 done: threading.Event = field(default_factory=threading.Event) 

301 value: Any = None 

302 ok: bool = False 

303 #: Set only when the owner's build raised an ordinary ``Exception`` — never 

304 #: for a ``BaseException`` that isn't one (`progress.Cancelled`, Streamlit's 

305 #: `StopException`). See `_shared_build`. 

306 error: Exception | None = None 

307 

308 

309#: UX-166: builds in progress, so a rerun that asks for the same frame waits 

310#: for the one already running instead of starting a second. A click during a 

311#: long load abandons the running script and starts a new one at once 

312#: (`runner.fastReruns`); without this the new run normalized the corpus again 

313#: beside the first. `st.cache_data` has the same guarantee through its own 

314#: per-key lock. 

315_INFLIGHT: dict[tuple, _InFlight] = {} 

316_INFLIGHT_LOCK = threading.Lock() 

317 

318#: `lookup`/`publish` (see `_shared_build`) return/accept this to mean "no 

319#: cached value" — never `None`, since a legitimate result can itself be `None`. 

320_MISSING = object() 

321 

322 

323def _shared_build( 

324 ident: tuple, 

325 build: Callable[[], Any], 

326 *, 

327 lookup: Callable[[], Any] | None = None, 

328 publish: Callable[[Any], None] | None = None, 

329) -> Any: 

330 """``build()``, run once for everyone asking for ``ident`` at the same time. 

331 

332 A caller that finds a build running waits for it and reuses its result. 

333 If the owner's build raises an ordinary ``Exception``, that same exception 

334 is re-raised in every waiter too — the input hasn't changed, so rebuilding 

335 would just fail again the same way. Only a ``BaseException`` that is *not* 

336 an ``Exception`` (`progress.Cancelled`, Streamlit's `StopException`) means 

337 nobody actually finished the build, so a waiter then builds it itself. 

338 

339 ``lookup``/``publish`` let a cache-shaped caller close UX-166's "latest 

340 request wins" race: a new owner calls ``lookup()`` right after winning 

341 ownership — a value another, faster build already published for this 

342 exact ``ident`` a moment earlier is reused without rebuilding — and a 

343 successful build calls ``publish(value)`` *before* the in-flight entry is 

344 popped, so "is this result still wanted, or has a newer request for this 

345 slot already been made" is decided while this ``ident`` still has exactly 

346 one owner. A joined waiter calls its own ``publish(entry.value)`` too 

347 (UX-166 fix-round-2, Minor #1 of Ruling T5-5): the owner's own decision was 

348 made against whatever key was latest *then*, and a request for this exact 

349 ``ident`` can itself become the latest again before the owner's entry is 

350 popped — without this, that waiter would still get the right *value* back 

351 but the store would never hold it. 

352 

353 Neither ``lookup`` nor ``publish`` is called for a plain (non-cache) use 

354 of this function, and neither's own failure is allowed to leak the 

355 in-flight entry or hang every waiter forever (UX-166 fix-round-2, Ruling 

356 T5-5 — a24e105 always popped the entry and signalled ``done``; the "latest 

357 request wins" fix lost that guarantee by writing the cleanup out per path 

358 instead of in one ``finally``): a failing ``lookup`` is treated as a miss 

359 (logged at debug, then built normally); a failing ``publish`` is logged as 

360 a warning and swallowed — the build itself already succeeded, ``entry.ok`` 

361 is already ``True``, and its caller still gets its value either way. 

362 """ 

363 while True: 

364 with _INFLIGHT_LOCK: 

365 entry = _INFLIGHT.get(ident) 

366 owner = entry is None 

367 if owner: 

368 entry = _InFlight() 

369 _INFLIGHT[ident] = entry 

370 if owner: 

371 # UX-166 fix-round-2 (Ruling T5-5): the whole owner branch is one 

372 # try/finally, so the registry pop and `done.set()` ALWAYS run — 

373 # whether `lookup`, `build` or `publish` raises, or nothing does. 

374 # Without this, a raising `publish` (the reviewer's repro: 

375 # `_register_fingerprints` racing another session's concurrent insert) 

376 # left the entry registered forever: every waiter already joined 

377 # blocks in `entry.done.wait()` with no timeout and no Streamlit 

378 # checkpoint to free it, and every later miss for this `ident` 

379 # joins the same dead entry and hangs too. 

380 try: 

381 hit = _MISSING 

382 if lookup is not None: 

383 try: 

384 hit = lookup() 

385 except Exception: 

386 _LOGGER.debug( 

387 "_shared_build lookup failed for %r; building instead", 

388 ident, 

389 exc_info=True, 

390 ) 

391 hit = _MISSING 

392 if hit is not _MISSING: 

393 value = hit 

394 else: 

395 try: 

396 value = build() 

397 except BaseException as exc: 

398 if isinstance(exc, Exception): 

399 entry.error = exc 

400 raise 

401 entry.value = value 

402 entry.ok = True 

403 # Only a build we actually ran gets published — a lookup hit 

404 # means the store already holds this exact key's value. 

405 if hit is _MISSING and publish is not None: 

406 try: 

407 publish(value) 

408 except Exception: 

409 # A WARNING, not DEBUG (the in-app log captures from 

410 # INFO): swallowed, a publish that keeps failing shows 

411 # only as every rerun rebuilding. 

412 _LOGGER.warning( 

413 "_shared_build publish failed for %r; the built " 

414 "value is still returned, just not cached", 

415 ident, 

416 exc_info=True, 

417 ) 

418 return value 

419 finally: 

420 with _INFLIGHT_LOCK: 

421 _INFLIGHT.pop(ident, None) 

422 entry.done.set() 

423 # UX-166: waiting on a build another run started is this run's work too 

424 # — the owner reports into its own task, so without this a gated card 

425 # over the wait (the Corpus measures, opened afresh each run) never 

426 # shows. It is also a cancel checkpoint: a waiter whose own task was 

427 # cancelled stops here instead of waiting out a build it no longer 

428 # wants. A hit returned above, so an all-hit rerun never gets here. 

429 progress.report() 

430 entry.done.wait() 

431 if entry.ok: 

432 if publish is not None: 

433 try: 

434 publish(entry.value) 

435 except Exception: 

436 _LOGGER.warning( 

437 "_shared_build waiter publish failed for %r", 

438 ident, 

439 exc_info=True, 

440 ) 

441 return entry.value 

442 if entry.error is not None: 

443 raise entry.error 

444 # The owner was cancelled or stopped, not merely wrong: nobody actually 

445 # built this. Loop back and become the new owner ourselves. 

446 

447 

448def frame_cache(slot: str, key, build, *, keep: int = 1): 

449 """Return ``build()``'s result, reusing the last one while ``key`` holds. 

450 

451 ``st.cache_data`` hands every caller a private **deep copy** of its result. 

452 That is the right default — it stops one part of the app corrupting 

453 another's data — but for the corpus-scale frames it buys nothing and costs a 

454 great deal: measured on the full OneStop reports, ~1.15 s and ~1.2 GB of 

455 allocate-and-discard on *every* rerun, i.e. on every widget touch, for data 

456 the app already had. 

457 

458 Nothing writes into those frames in place. That is not an assumption: 

459 ``tests/test_frame_immutability.py`` captures them as the loader yields them, 

460 runs a full render and a rerun on top, and asserts they come back 

461 byte-identical — and the same file's canary proves the check would notice if 

462 they didn't. So this hands back **the object itself**. 

463 

464 One entry per slot by default, deliberately: keeping the previous corpus 

465 alive beside the current one costs memory. ``keep`` raises that for a slot 

466 where going back is the common move (PERF-18: the normalized pair keeps 

467 two, so switching PoTeC → OneStop → PoTeC doesn't normalize PoTeC again — 

468 a ~20 s wait at OneStop scale). Falls back to 

469 calling ``build`` when there is no session state, which is what the headless 

470 API and the CLI see. 

471 

472 Keys must be hashable (they are compared with ``==`` and stored as dict 

473 keys). Concurrent requests for the same slot + key share one build in 

474 flight (`_shared_build`); a build that finishes only *publishes* — writes 

475 the entry every later request for that key reuses — while its key is still 

476 the slot's most recently requested one (UX-166's "latest request wins"), 

477 so a superseded build's late finish can never clobber a newer result. It 

478 still returns its value to its own caller either way. 

479 """ 

480 try: 

481 store = st.session_state.setdefault(_FRAME_CACHE_KEY, {}) 

482 except (RuntimeError, AttributeError, KeyError) as exc: 

483 # No Streamlit runtime (api.py, cli.py, a bare import) — nothing to 

484 # cache into, and nothing that reruns. Logged rather than silent: the 

485 # symptom of catching this wrongly is a cache that simply never works, 

486 # which is invisible except as everything being slow. 

487 _LOGGER.debug("frame_cache falling back to a plain call: %s", exc) 

488 return build() 

489 # UX-166: record this as the slot's latest request on EVERY call — a hit 

490 # included, since "the user cancelled back to an earlier dataset" is a hit 

491 # for the slot's *current* entry, and without recording it here too an 

492 # abandoned build for a *different* key would still look, to its own late 

493 # `publish`, like nobody had asked for anything else since. 

494 with _INFLIGHT_LOCK: 

495 store[(_LATEST_REQUESTED, slot)] = key 

496 entry = store.get(slot) 

497 if entry is not None and entry[0] == key: 

498 return entry[1] 

499 if keep > 1: 

500 with _INFLIGHT_LOCK: 

501 earlier = store.get((_EARLIER, slot)) or [] 

502 for index, (old_key, old_value) in enumerate(earlier): 

503 if old_key == key: 

504 # Promote it back to current; the entry it replaces 

505 # becomes the most recent earlier one. 

506 rest = earlier[:index] + earlier[index + 1 :] 

507 current = store.get(slot) 

508 store[(_EARLIER, slot)] = ([current] if current else []) + rest 

509 store[slot] = (old_key, old_value) 

510 return old_value 

511 

512 def _lookup() -> Any: 

513 # UX-166: a new owner re-checks the store before building — a 

514 # concurrent build for this exact key may have just published, 

515 # between our own miss above and winning ownership below. 

516 with _INFLIGHT_LOCK: 

517 current = store.get(slot) 

518 if current is not None and current[0] == key: 

519 return current[1] 

520 return _MISSING 

521 

522 def _publish(value: Any) -> None: 

523 with _INFLIGHT_LOCK: 

524 wins = store.get((_LATEST_REQUESTED, slot)) == key 

525 if wins: 

526 previous = store.get(slot) 

527 if keep > 1 and previous is not None and previous[0] != key: 

528 earlier = [ 

529 item 

530 for item in store.get((_EARLIER, slot)) or [] 

531 if item[0] != key 

532 ] 

533 store[(_EARLIER, slot)] = [previous, *earlier][: keep - 1] 

534 store[slot] = (key, value) 

535 if wins: 

536 # BUG-103: the key decides the value, so it is the value's ID too. 

537 _register_fingerprints( 

538 [ 

539 (frame, ("assigned", ("frame_cache", slot, key, label))) 

540 for label, frame in _frame_parts(value) 

541 ] 

542 ) 

543 

544 # UX-166: shared with a build already running for this session, slot and key. 

545 return _shared_build( 

546 (id(store), slot, key), build, lookup=_lookup, publish=_publish 

547 ) 

548 

549 

550def clear_frame_cache() -> None: 

551 """Drop every no-copy frame cache entry (PERF-6). 

552 

553 Called by ``app.clear_computation_cache`` alongside ``st.cache_data.clear``: 

554 the normalized frames are cached *here* rather than there, so clearing only 

555 Streamlit's cache would leave a just-deleted dataset's frames alive. 

556 """ 

557 try: 

558 st.session_state.pop(_FRAME_CACHE_KEY, None) 

559 except (RuntimeError, AttributeError, KeyError): 

560 pass # No runtime: there is no cache to clear. 

561 

562 

563def reset_fingerprint_memo() -> None: 

564 """Drop the per-run fingerprint memo. Called once per script run (PERF-3).""" 

565 getattr(_FINGERPRINT_MEMO, "cache", {}).clear() 

566 

567 

568def frame_fingerprint(df: pd.DataFrame | None) -> tuple: 

569 """Exact, content-sensitive identity for a DataFrame. 

570 

571 Used as an *explicit* ``@st.cache_data`` key for functions that take an 

572 underscore-prefixed (un-hashed) frame argument — so Streamlit never re-hashes 

573 a multi-million-row frame on every rerun just to look up the cache. 

574 

575 **Two frames share a fingerprint only when they are the same data** 

576 (BUG-103). It is one of: 

577 

578 * the ID the app *assigned* the frame — from the loader that read it, the 

579 ``frame_cache`` entry that holds it, or the step that derived it (see 

580 ``_STABLE_FINGERPRINTS``). That is what keeps a corpus-sized frame from 

581 being hashed on every rerun; 

582 * otherwise ``(rows, columns, digest)`` over **every** row. Until BUG-103 a 

583 frame over 200,000 rows was keyed on ~384 sampled rows, so a corrected 

584 re-upload of the same shape matched the old file and was served its 

585 results — and its normalized tables, which were then stored as the new 

586 dataset — without a word. A full hash costs ~60 ms per million rows, which 

587 is why the large frames on the rerun path are assigned an ID instead. 

588 

589 The per-row hashes are digested **in order** rather than summed. Summing is 

590 order-invariant, so a frame and a ``sort_values`` of itself — same rows, same 

591 index labels, different order — used to share a key while producing different 

592 results downstream. 

593 

594 ``hash_pandas_object`` raises on columns of unhashable objects (lists/arrays — 

595 e.g. parquet-preserved span-index fields). We stringify and retry rather than 

596 drop the content signal entirely: zeroing the hash would collapse every frame 

597 of the same shape + columns to one fingerprint, serving stale cached results 

598 when switching between two such frames. If even that fails the key becomes 

599 *unique* rather than zero — an unhashable frame must miss the cache, not 

600 match every other frame of its shape. 

601 

602 **The same frame object is only hashed once per run** — see the 

603 ``_FINGERPRINT_MEMO`` note above for why that is safe and what it assumes. 

604 """ 

605 if df is None or getattr(df, "empty", True): 

606 return (0, ()) 

607 memo = getattr(_FINGERPRINT_MEMO, "cache", None) 

608 if memo is None: 

609 memo = _FINGERPRINT_MEMO.cache = OrderedDict() 

610 key = id(df) 

611 hit = memo.get(key) 

612 if hit is not None and hit[0]() is df: 

613 memo.move_to_end(key) 

614 return hit[1] 

615 stable = _STABLE_FINGERPRINTS.get(key) 

616 if stable is not None and stable[0]() is not df: 

617 stable = None 

618 if stable is not None and stable[1] is not None: 

619 return stable[1] 

620 value = _compute_frame_fingerprint(df) 

621 if stable is not None: 

622 with _STABLE_FINGERPRINTS_LOCK: 

623 _STABLE_FINGERPRINTS[key] = (stable[0], value) 

624 # Drop entries whose frame has already been collected before evicting a live 

625 # one — those are pure bookkeeping and cost nothing to lose. 

626 if len(memo) >= _FINGERPRINT_MEMO_MAX: 

627 for dead in [k for k, (ref, _) in memo.items() if ref() is None]: 

628 del memo[dead] 

629 while len(memo) >= _FINGERPRINT_MEMO_MAX: 

630 memo.popitem(last=False) 

631 memo[key] = (weakref.ref(df), value) 

632 return value 

633 

634 

635def _compute_frame_fingerprint(df: pd.DataFrame) -> tuple: 

636 """The actual hash behind :func:`frame_fingerprint`, memo aside.""" 

637 cols = tuple(map(str, df.columns)) 

638 n = len(df) 

639 

640 def _hash(frame: pd.DataFrame) -> str: 

641 try: 

642 per_row = pd.util.hash_pandas_object(frame, index=True) 

643 except TypeError: 

644 # Unhashable cell objects — stringify so content still drives the key. 

645 per_row = pd.util.hash_pandas_object(frame.astype(str), index=True) 

646 # Digest the per-row hashes in ORDER. Summing them (the previous 

647 # approach) is order-invariant, so a frame and a `sort_values` of itself 

648 # — same rows, same index labels, different order — produced the same 

649 # key while yielding different results downstream. 

650 return hashlib.blake2b( 

651 per_row.to_numpy(dtype="uint64").tobytes(), digest_size=16 

652 ).hexdigest() 

653 

654 try: 

655 return (n, cols, _hash(df)) 

656 except Exception: 

657 # Fail CLOSED: a key nothing else can equal, so this frame simply doesn't 

658 # share a cache entry. `(n, cols, 0, 0)` failed *open* — every frame of 

659 # the same shape collided. 

660 return (n, cols, uuid.uuid4().hex) 

661 

662 

663# --------------------------------------------------------------------------- 

664# Server-side OneStop data source. 

665# 

666# When the env var `ONESTOP_DATA_DIR` points at a OneStop lacclab export 

667# folder (containing `ia_Paragraph.csv.zip` and `fixations_Paragraph.csv.zip`), 

668# `load_onestop_server_bundle()` returns them as the (words, fixations) tuple 

669# the rest of the pipeline expects. The schema is identical to the bundled 

670# sample (the sample is a 3-pid subset of OneStop), so no extra normalisation 

671# is required. 

672# 

673# Drives the "OneStop server bundle" data source option in app.py, used by 

674# an external review-app deep-link integration (single pid+trial into this UI). 

675# --------------------------------------------------------------------------- 

676 

677ONESTOP_DATA_DIR_ENV = "ONESTOP_DATA_DIR" 

678 

679 

680def onestop_data_dir() -> Path | None: 

681 """Resolved value of `$ONESTOP_DATA_DIR`, or `None` if unset/blank.""" 

682 raw = os.environ.get(ONESTOP_DATA_DIR_ENV, "").strip() 

683 return Path(raw) if raw else None 

684 

685 

686def onestop_full_bundle_exists() -> bool: 

687 """True when the full OneStop CSV.zip exports are present (not just per-pid 

688 shards). 

689 

690 When the whole corpus is available the app loads it once and filters in-app 

691 (so switching participant is instant); a shards-only setup must instead load 

692 one participant's shard at a time (it can't materialize the ~60 GB corpus).""" 

693 base = onestop_data_dir() 

694 if base is None: 

695 return False 

696 return (base / "ia_Paragraph.csv.zip").exists() and ( 

697 base / "fixations_Paragraph.csv.zip" 

698 ).exists() 

699 

700 

701def _onestop_shard_paths(base: Path, pid: str) -> tuple[Path, Path]: 

702 """Resolved per-participant shard paths under `<base>/by_pid/`.""" 

703 pid = pid.strip().lower() 

704 return ( 

705 base / "by_pid" / "ia" / f"{pid}.parquet", 

706 base / "by_pid" / "fixations" / f"{pid}.parquet", 

707 ) 

708 

709 

710def onestop_data_provenance(participant: str | None = None) -> dict: 

711 """Where the currently-loaded OneStop data came from, for the Raw Data tab. 

712 

713 Parses `ONESTOP_DATA_DIR` (typically `…/onestop_<cohort>/reports/<source>/<date>/full/`) 

714 to surface cohort, export source (lacclab / public / osf), and date in the 

715 UI so reviewers can verify they're looking at the right export. Also 

716 reports the per-pid shard's mtime when a participant is set — that's the 

717 timestamp of the actual data the page is currently rendering. 

718 

719 Returns an empty dict when `ONESTOP_DATA_DIR` is unset (i.e. the OneStop 

720 data source isn't in use — caller should suppress the provenance panel). 

721 """ 

722 base = onestop_data_dir() 

723 if base is None: 

724 return {} 

725 

726 info: dict = {"data_dir": str(base)} 

727 

728 # Best-effort parse of the canonical path layout. 

729 parts = base.resolve().parts 

730 try: 

731 # Look for "reports" anchor and grab source/date after it. 

732 i = parts.index("reports") 

733 info["source"] = parts[i + 1] # lacclab / public / osf 

734 info["date"] = parts[i + 2] # YYYYMMDD 

735 except (ValueError, IndexError): 

736 pass 

737 for p in parts: 

738 if p.startswith("onestop_"): 

739 info["cohort"] = p.removeprefix("onestop_") # L1 / L2 

740 break 

741 

742 # Reports the per-pid shard's mtime when a participant is set — that's the 

743 # timestamp of the bytes the page is rendering right now. 

744 if participant: 

745 ia_shard, fix_shard = _onestop_shard_paths(base, participant) 

746 info["loaded_from"] = "per-pid shard" 

747 info["ia_shard"] = str(ia_shard) 

748 info["fix_shard"] = str(fix_shard) 

749 if ia_shard.is_file(): 

750 info["ia_shard_mtime"] = ia_shard.stat().st_mtime 

751 if fix_shard.is_file(): 

752 info["fix_shard_mtime"] = fix_shard.stat().st_mtime 

753 else: 

754 ia_csv = base / "ia_Paragraph.csv.zip" 

755 fix_csv = base / "fixations_Paragraph.csv.zip" 

756 info["loaded_from"] = "full CSV.zip export" 

757 if ia_csv.is_file(): 

758 info["ia_shard"] = str(ia_csv) 

759 info["ia_shard_mtime"] = ia_csv.stat().st_mtime 

760 if fix_csv.is_file(): 

761 info["fix_shard"] = str(fix_csv) 

762 info["fix_shard_mtime"] = fix_csv.stat().st_mtime 

763 return info 

764 

765 

766# UX-166: the dataset card lists this step. 

767@st.cache_data(show_spinner=False) 

768def load_onestop_server_bundle( 

769 participant: str | None = None, 

770) -> tuple[pd.DataFrame, pd.DataFrame]: 

771 """Load OneStop lacclab IA + fixation reports from `$ONESTOP_DATA_DIR`. 

772 

773 Fast path — when `participant` is given and per-pid shards exist under 

774 `<ONESTOP_DATA_DIR>/by_pid/{ia,fixations}/<pid>.parquet`, load just that 

775 one participant (sub-second). Shards are generated by 

776 `python -m scanpath_studio.onestop_shard --data-dir <ONESTOP_DATA_DIR>`. 

777 

778 Slow path — fall back to loading the full CSV.zip exports (~3 min, ~60 GB 

779 RAM for the L2 cohort). Used when no participant is specified, or when 

780 a deep link points at a pid whose shard hasn't been generated yet. 

781 """ 

782 progress.report() # UX-166: a miss — real work, so the gated card may show 

783 base = onestop_data_dir() 

784 if base is None: 

785 return pd.DataFrame(), pd.DataFrame() 

786 

787 # Fast path: per-pid shards. 

788 if participant: 

789 ia_shard, fix_shard = _onestop_shard_paths(base, participant) 

790 ia_present = ia_shard.exists() 

791 fix_present = fix_shard.exists() 

792 if ia_present and fix_present: 

793 # PERF-6: parse only the columns the mapping + registry keep. 

794 return stamp_source( 

795 ( 

796 read_mapped_table(ia_shard, kind="words"), 

797 read_mapped_table(fix_shard, kind="fixations"), 

798 ) 

799 ) 

800 # NEVER fall through to the 15 GB load when a participant is named — 

801 # the deep link is for one pid only, so loading the whole cohort just 

802 # to discover the pid still has no data is pure waste. Surface a clear 

803 # error and stop. Common cause: pid was excluded from the IA report 

804 # (no exported reading data), or shards haven't been generated yet. 

805 missing = [ 

806 f"{p.parent.name}/{p.name}" 

807 for p, ok in [(ia_shard, ia_present), (fix_shard, fix_present)] 

808 if not ok 

809 ] 

810 st.error( 

811 f"No data for participant {participant!r} on this server: they have " 

812 f"no reading data, or its files ({', '.join(missing)}) are missing. " 

813 "Whoever runs the server can regenerate them with " 

814 "`python -m scanpath_studio.onestop_shard --data-dir <ONESTOP_DATA_DIR>`." 

815 ) 

816 st.stop() 

817 

818 # Slow path: full CSV.zip load. 

819 ia_path = base / "ia_Paragraph.csv.zip" 

820 fix_path = base / "fixations_Paragraph.csv.zip" 

821 if not ia_path.exists() or not fix_path.exists(): 

822 st.error( 

823 f"OneStop data not found under {base}. Expected ia_Paragraph.csv.zip + " 

824 f"fixations_Paragraph.csv.zip." 

825 ) 

826 return pd.DataFrame(), pd.DataFrame() 

827 # PERF-6: an IA report ships 174 columns and a fixation report 300, of 

828 # which normalization keeps 27 and 17. Parsing the rest is what made this 

829 # path reach ~25 GB resident before a single measure was computed. 

830 words = read_mapped_table(ia_path, kind="words") 

831 fixations = read_mapped_table(fix_path, kind="fixations") 

832 return stamp_source((words, fixations)) 

833 

834 

835_TRAILING_UNIT = re.compile(r"\s*[\[(][^\[\]()]*[\])]\s*$") 

836 

837 

838def _norm_col(name) -> str: 

839 """Fold a column name to its case- and separator-insensitive key. 

840 

841 Lowercases and drops every non-alphanumeric char, so ``IA_LEFT``, 

842 ``ia_left``, ``Ia-Left`` and ``ia left`` all collapse to ``ialeft`` — 

843 letting auto-detection match real-world column names that differ only in 

844 capitalization or word separators. 

845 

846 A trailing unit or coordinate-system block is dropped first (DATA-25): 

847 Tobii Pro Lab writes ``Fixation point X [DACS px]``, SMI BeGaze 

848 ``Fixation Duration [ms]``, Pupil Labs Neon ``fixation x [px]`` and Tobii 

849 Studio ``FixationPointX (MCSpx)``. Folding the brackets' *letters* into the 

850 key (``fixationpointxdacspx``) made every one of them miss a candidate that 

851 names the same thing, so the export of three major vendors landed in the 

852 manual mapping step. Only one trailing block goes — a bracket in the middle 

853 of a name is part of the name.""" 

854 text = _TRAILING_UNIT.sub("", str(name)) 

855 return re.sub(r"[^a-z0-9]", "", text.lower()) 

856 

857 

858_COL_SEPARATORS = re.compile(r"[^a-zA-Z0-9]+") 

859_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])") 

860 

861 

862def _col_tokens(name) -> list[str]: 

863 """Split a raw column name into its separator-delimited tokens (DATA-25 

864 second pass), each folded like ``_norm_col``. 

865 

866 The trailing-unit block is dropped from the *whole* name first, same as 

867 ``_norm_col`` — so a vendor's ``LEFT_px`` tokenizes to ``["left", "px"]`` 

868 with the unit noise already gone, not left to coincidentally never match 

869 a candidate.""" 

870 text = _TRAILING_UNIT.sub("", str(name)) 

871 # DATA-57: a CamelCase name (`BoxLeft`, `AoiTop`) has no separator to split 

872 # on, so a lower→upper case change counts as one. 

873 text = _CAMEL_BOUNDARY.sub(" ", text) 

874 return [tok.lower() for tok in _COL_SEPARATORS.split(text) if tok] 

875 

876 

877def pick_column(df: pd.DataFrame, candidates: Iterable[str]) -> str | None: 

878 """Return the first matching column name from a candidate list. 

879 

880 Matching is case- and separator-insensitive (see ``_norm_col``). Candidate 

881 order is still priority order — the first candidate with any match wins (so 

882 EyeLink names keep beating Gazepoint), and among equally-normalized columns 

883 the leftmost one wins. 

884 

885 If nothing matches exactly, a second pass (DATA-25) catches a vendor 

886 prefix or suffix on a known name — ``AOI_LEFT``, ``LEFT_px`` — by 

887 splitting each column on its separators and checking whether any *whole* 

888 token equals a candidate. There is no prefix vocabulary to maintain, and 

889 no substring matching, so ``top`` never matches ``stop_time`` (the whole 

890 token is ``stop``, not ``top``) and ``id`` never matches ``guid``. That 

891 still leaves real ambiguity — ``top_left_x`` and ``top_left_y`` both 

892 contain the token ``left``; ``max_x`` and ``fix_x`` both contain ``x`` — 

893 so the second pass is accepted only when it turns up **exactly one** 

894 column across every candidate in the list. Two or more survivors is 

895 ambiguity, and ambiguity means the manual mapping step, not a guess: the 

896 safety here is uniqueness, not a whitelist. 

897 

898 Several survivors get one narrowing step (DATA-60) before that verdict: 

899 keep only the ones spelled **entirely** in the list's own words — every 

900 token of the column is a token of some candidate. An AOI export's 

901 ``AOI_ID`` beside ``AOI_LEFT`` … ``AOI_BOTTOM`` is the case: all five carry 

902 the word-id candidate ``aoi``, but only ``AOI_ID``'s other token (``id``, 

903 from ``word_id`` / ``IA_ID``) belongs to the word-id list, while ``left`` 

904 does not. The same uniqueness applies to what is left: one column, or 

905 none.""" 

906 lookup: dict[str, str] = {} 

907 for col in df.columns: 

908 lookup.setdefault(_norm_col(col), col) 

909 candidates = list(candidates) 

910 for name in candidates: 

911 hit = lookup.get(_norm_col(name)) 

912 if hit is not None: 

913 return hit 

914 

915 normed_candidates = {_norm_col(name) for name in candidates} 

916 survivors = [col for col in df.columns if normed_candidates & set(_col_tokens(col))] 

917 if len(survivors) > 1: 

918 vocabulary = {tok for name in candidates for tok in _col_tokens(name)} 

919 survivors = [col for col in survivors if set(_col_tokens(col)) <= vocabulary] 

920 if len(survivors) == 1: 

921 return survivors[0] 

922 return None 

923 

924 

925#: Milliseconds per unit, for a time column whose header names its unit — 

926#: Tobii Pro Lab's `Recording timestamp [μs]`, Pupil Labs Neon's 

927#: `start timestamp [ns]` (DATA-40). DATA-25 taught auto-detection to look past 

928#: that block; this is what reads it. 

929_TIME_UNIT_MS = { 

930 "s": 1000.0, 

931 "sec": 1000.0, 

932 "secs": 1000.0, 

933 "seconds": 1000.0, 

934 "ms": 1.0, 

935 "msec": 1.0, 

936 "milliseconds": 1.0, 

937 "us": 1e-3, 

938 "µs": 1e-3, # MICRO SIGN 

939 "μs": 1e-3, # GREEK SMALL LETTER MU — what Tobii writes 

940 "microseconds": 1e-3, 

941 "ns": 1e-6, 

942 "nanoseconds": 1e-6, 

943} 

944#: Vendor time columns whose unit is in the manual, not the header: Gazepoint's 

945#: fixation start/duration and Pupil Labs Core's fixation onset are seconds. 

946_SECONDS_WITHOUT_A_SUFFIX = frozenset({"fpogd", "fpogs", "starttimestamp"}) 

947 

948 

949def time_unit_ms(column) -> float: 

950 """Milliseconds per unit of the time column named ``column`` (DATA-40). 

951 

952 Read from the header's trailing unit block (``[s]``, ``(ns)``, ``[μs]``) or, 

953 for a vendor column that carries none, from its documented unit. Anything 

954 else — no block, ``[ms]``, a block that names no time unit — is 1: the app's 

955 unit, and the only safe guess. 

956 """ 

957 match = _TRAILING_UNIT.search(str(column)) 

958 if match: 

959 unit = match.group(0).strip().strip("[]()").strip().lower() 

960 return _TIME_UNIT_MS.get(unit, 1.0) 

961 return 1000.0 if _norm_col(column) in _SECONDS_WITHOUT_A_SUFFIX else 1.0 

962 

963 

964def _as_ms(values: pd.Series, column) -> pd.Series: 

965 """``values`` of the time column ``column`` converted to milliseconds.""" 

966 factor = time_unit_ms(column) 

967 return values if factor == 1.0 else values * factor 

968 

969 

970def trial_mapping_columns(trial_mapping) -> list: 

971 """Column list behind a trial mapping — a plain column name or a list of 

972 names (the column-mapping UI returns a list when the user composes a 

973 unique trial ID from several columns).""" 

974 if isinstance(trial_mapping, str): 

975 return [trial_mapping] 

976 return list(trial_mapping) 

977 

978 

979#: Matches an otherwise-integer string with a spurious trailing ``.0`` — 

980#: exactly what a whole-number id column becomes once pandas has any reason to 

981#: read it as ``float64`` instead of ``int64``. 

982_WHOLE_FLOAT_ID = re.compile(r"^(-?\d+)\.0$") 

983 

984 

985def stable_id(series: pd.Series) -> pd.Series: 

986 """Cast an id column to the string every identity join compares it by. 

987 

988 Plain ``.astype(str)`` spells the *same* id two ways when one file's id 

989 column is read as whole numbers and another's is read as decimals — and it 

990 takes only one blank cell anywhere in an otherwise-integer CSV column to 

991 flip the whole column from ``int64`` to ``float64``. A trial's AOI table 

992 and its fixations report almost always come from two different exports, so 

993 this is not a rare corpus quirk: it is common for exactly one of the two to 

994 have that one blank cell the other does not. The trial picker, every 

995 metadata join (DATA-20/DATA-29), fixation-to-word assignment, and every 

996 filter then compare ``"101"`` against ``"101.0"`` as strings and find no 

997 match, even though both name the same trial. 

998 

999 Dropping a trailing ``.0`` off an otherwise-integer string is the one 

1000 collapse worth making here, and only where the ``.0`` is how *numbers* were 

1001 written (round 10): a column read as decimals, a float cell in a mixed 

1002 column, or a text column whose every whole number carries ``.0`` (an id 

1003 column a script wrote out as floats). A text column that spells some whole 

1004 numbers with ``.0`` and some without — ``"1"`` beside ``"1.0"`` — holds 

1005 opaque ids, and every spelling in it stays as written. The rule reads the 

1006 column's shape, not whether a ``"1"`` happens to sit beside a ``"1.0"``, so 

1007 two tables of the same shape decide alike. 

1008 """ 

1009 text = series.astype(str).str.strip() 

1010 # A full match and a slice rather than a backreferenced `re.sub`, which 

1011 # pandas runs per element in Python (5x slower per million rows). 

1012 pointed = text.str.fullmatch(_WHOLE_FLOAT_ID).fillna(False).astype(bool) 

1013 if not pointed.any(): 

1014 return text 

1015 collapsed = text.mask(pointed, text.str.slice(stop=-2)) 

1016 if pd.api.types.is_float_dtype(series): 

1017 return collapsed 

1018 whole = text.str.fullmatch(r"-?\d+(?:\.0)?").fillna(False).astype(bool) 

1019 if bool(pointed[whole].all()): 

1020 return collapsed # every whole number written as a float 

1021 if series.dtype != object: 

1022 return text 

1023 # A mixed object column: a real float cell is a number, the rest are text. 

1024 # By position, never by label — a concatenated frame repeats its labels. 

1025 mask = pointed.to_numpy() 

1026 keep = mask.copy() 

1027 keep[mask] = [not isinstance(value, float) for value in series.to_numpy()[mask]] 

1028 return collapsed.mask(keep, text) 

1029 

1030 

1031_DIGITS_ONLY = re.compile(r"^\d+$") 

1032 

1033 

1034def zero_padding_map(ids: Iterable, reference: Iterable) -> dict[str, str]: 

1035 """How ``ids`` would be spelled in ``reference``, when the only thing 

1036 keeping the two apart is zero-padding (BUG-59). 

1037 

1038 One table read ``007`` as text and another read it as the number 7, and 

1039 every join between them then matched nothing — words to fixations, a 

1040 participant table to the data. Returns ``{"7": "007", …}`` for each id in 

1041 ``ids`` whose zero-padded twin is in ``reference``, and ``{}`` whenever that 

1042 is not the *only* story: if any id already matches as it is, if either side 

1043 has two ids that differ only by padding (``1`` and ``01`` — genuinely 

1044 different ids), or if the padded spelling is not all on one side. Nothing is 

1045 renamed on a guess. 

1046 """ 

1047 own = {str(v) for v in ids if pd.notna(v)} 

1048 other = {str(v) for v in reference if pd.notna(v)} 

1049 if not own or not other or own & other: 

1050 return {} 

1051 

1052 def by_value(values: set) -> dict | None: 

1053 keyed: dict = {} 

1054 for value in values: 

1055 if _DIGITS_ONLY.match(value): 

1056 key = value.lstrip("0") or "0" 

1057 if key in keyed: 

1058 return None 

1059 keyed[key] = value 

1060 return keyed 

1061 

1062 mine, theirs = by_value(own), by_value(other) 

1063 if mine is None or theirs is None: 

1064 return {} 

1065 shared = mine.keys() & theirs.keys() 

1066 if not shared or any(len(mine[k]) >= len(theirs[k]) for k in shared): 

1067 return {} 

1068 return {mine[k]: theirs[k] for k in shared} 

1069 

1070 

1071#: The separator between the parts of a composite id, and the escape that lets a 

1072#: part contain it. See :func:`compose_id`. 

1073COMPOSITE_SEPARATOR = "_" 

1074_COMPOSITE_ESCAPE = "\\" 

1075 

1076 

1077def _escape_parts(parts: pd.Series) -> pd.Series: 

1078 return parts.str.replace( 

1079 _COMPOSITE_ESCAPE, _COMPOSITE_ESCAPE * 2, regex=False 

1080 ).str.replace( 

1081 COMPOSITE_SEPARATOR, _COMPOSITE_ESCAPE + COMPOSITE_SEPARATOR, regex=False 

1082 ) 

1083 

1084 

1085def compose_id(parts: Iterable) -> str: 

1086 r"""One composite id from its parts, such that different parts never give 

1087 the same id. 

1088 

1089 The parts are joined with ``_``. A part that itself contains ``_`` or ``\`` 

1090 has each one escaped with a ``\`` first, so ``("block_A", "B")`` is 

1091 ``block\_A_B`` and ``("block", "A_B")`` is ``block_A\_B`` — before the 

1092 escape both were ``block_A_B``, and two readings became one trial. Parts 

1093 with neither character, which is almost every id, compose exactly as they 

1094 always did (``("p1", "t3")`` is still ``p1_t3``). The encoding is 

1095 reversible, which is what makes it injective: :func:`split_composite_id` 

1096 reads the parts back. :func:`trial_id_series` is the vectorised form. 

1097 """ 

1098 escaped = _escape_parts(pd.Series([str(part) for part in parts], dtype=object)) 

1099 return COMPOSITE_SEPARATOR.join(escaped) 

1100 

1101 

1102def split_composite_id(value: str) -> list[str]: 

1103 """The parts :func:`compose_id` joined into ``value``.""" 

1104 parts: list[str] = [] 

1105 current: list[str] = [] 

1106 chars = iter(str(value)) 

1107 for char in chars: 

1108 if char == _COMPOSITE_ESCAPE: 

1109 current.append(next(chars, _COMPOSITE_ESCAPE)) 

1110 elif char == COMPOSITE_SEPARATOR: 

1111 parts.append("".join(current)) 

1112 current = [] 

1113 else: 

1114 current.append(char) 

1115 parts.append("".join(current)) 

1116 return parts 

1117 

1118 

1119def legacy_composite_id(value: str) -> str: 

1120 """How a composite id was spelled before :func:`compose_id` escaped its 

1121 parts — plainly joined with ``_``. The same as ``value`` unless one of its 

1122 parts held a ``_`` or a backslash.""" 

1123 value = str(value) 

1124 if _COMPOSITE_ESCAPE not in value: 

1125 return value 

1126 return COMPOSITE_SEPARATOR.join(split_composite_id(value)) 

1127 

1128 

1129def composite_respelling_map(ids: Iterable, reference: Iterable) -> dict[str, str]: 

1130 r"""How each id in ``ids`` is spelled in ``reference``, when the only thing 

1131 keeping them apart is the escaping :func:`compose_id` added to composite ids. 

1132 

1133 An id saved before it — in a stored dataset, an annotations file, a link — 

1134 spells a composite id whose parts contain ``_`` without the escapes; one 

1135 composed since spells it with them. Returns ``{"block_A_B": "block\_A_B"}`` 

1136 (or the reverse), for the ids of ``ids`` that ``reference`` lacks. An old 

1137 spelling two current ids share is left out: that old id named both readings 

1138 at once, and nothing here picks one. Nothing is renamed on a guess, as with 

1139 :func:`zero_padding_map`. 

1140 """ 

1141 own = {str(v) for v in ids if pd.notna(v)} 

1142 other = {str(v) for v in reference if pd.notna(v)} 

1143 missing = own - other 

1144 if not missing or not other: 

1145 return {} 

1146 by_legacy: dict[str, str | None] = {} 

1147 for value in other: 

1148 legacy = legacy_composite_id(value) 

1149 if legacy != value: 

1150 by_legacy[legacy] = None if legacy in by_legacy else value 

1151 mapping: dict[str, str] = {} 

1152 for value in missing: 

1153 current = by_legacy.get(value) 

1154 if current is not None: 

1155 mapping[value] = current 

1156 continue 

1157 legacy = legacy_composite_id(value) 

1158 if legacy != value and legacy in other: 

1159 mapping[value] = legacy 

1160 # Two ids landing on one would merge them — leave both alone. 

1161 landed: dict[str, int] = {} 

1162 for target in mapping.values(): 

1163 landed[target] = landed.get(target, 0) + 1 

1164 return {k: v for k, v in mapping.items() if landed[v] == 1} 

1165 

1166 

1167def respell_reading(participant, trial, readings: Iterable) -> tuple[str, str]: 

1168 """``(participant, trial)`` spelled the way ``readings`` — ``(participant, 

1169 trial)`` pairs — spell them, through :func:`composite_respelling_map`. 

1170 

1171 For an id saved before composite ids escaped a ``_`` inside a part: a link, 

1172 an annotations file or a script. Each half is respelled only when it is 

1173 missing as given and its other spelling is unambiguous; otherwise it comes 

1174 back unchanged and the caller's own "not found" applies. 

1175 """ 

1176 pid, tid = str(participant), str(trial) 

1177 pairs = ( 

1178 readings 

1179 if isinstance(readings, frozenset) # already strings, e.g. a trial set 

1180 else frozenset((str(p), str(t)) for p, t in readings) 

1181 ) 

1182 if (pid, tid) in pairs: 

1183 return pid, tid 

1184 pid = composite_respelling_map([pid], {p for p, _ in pairs}).get(pid, pid) 

1185 own = {t for p, t in pairs if p == pid} or {t for _, t in pairs} 

1186 return pid, composite_respelling_map([tid], own).get(tid, tid) 

1187 

1188 

1189def trial_id_series(source: pd.DataFrame, trial_mapping) -> pd.Series: 

1190 """Trial-id values for a single-column or composite (multi-column) mapping. 

1191 

1192 A multi-column mapping builds a unique trial ID on the fly from the 

1193 columns' string values with :func:`compose_id` — joined with ``_``, a ``_`` 

1194 or backslash inside a part escaped, so two different tuples never share an 

1195 id — for datasets that ship no precomputed unique-trial column (e.g. 

1196 OneStop-style participant + paragraph + repeated-reading). Each component 

1197 is passed through :func:`stable_id` first, so a composite id cannot inherit 

1198 a ``.0`` from one of its parts. 

1199 """ 

1200 cols = trial_mapping_columns(trial_mapping) 

1201 if len(cols) == 1: 

1202 return stable_id(source[cols[0]]) 

1203 escaped = [_escape_parts(stable_id(source[c])) for c in cols] 

1204 return escaped[0].str.cat(escaped[1:], sep=COMPOSITE_SEPARATOR) 

1205 

1206 

1207def _preserve_composite_columns( 

1208 df: pd.DataFrame, source: pd.DataFrame, trial_mapping 

1209) -> pd.DataFrame: 

1210 """Carry a composite mapping's source columns into the normalized frame 

1211 under their original names. 

1212 

1213 A multi-column trial mapping gets joined into a single opaque ``trial_id`` 

1214 (e.g. ``2_1_1_Ele_l37_1129_False``). Keeping the individual component columns 

1215 is what lets the trial chips spell that id back out part by part 

1216 (``tabs._render_trial_condition_chips``). It no longer changes how the trial 

1217 is *picked* — BUG-23 made the picker the same for every mapping. 

1218 No-op for single-column mappings. 

1219 Rows are 1:1 with ``source`` here (no filtering in the composite path), so a 

1220 positional copy stays aligned.""" 

1221 cols = trial_mapping_columns(trial_mapping) 

1222 if len(cols) < 2: 

1223 return df 

1224 for col in cols: 

1225 if col not in df.columns and col in source.columns: 

1226 df[col] = source[col].to_numpy() 

1227 return df 

1228 

1229 

1230# Candidate column names checked during auto-inference. Centralised so the 

1231# proposal step and the override UI share the same defaults. Matching is case- 

1232# and separator-insensitive (see ``pick_column``), so these list only *distinct* 

1233# conventions — no ALL_CAPS / snake_case twins of the same name needed. 

1234PARTICIPANT_CANDIDATES = [ 

1235 "participant_id", 

1236 "unique_participant_id", # EyeGenBench's own harmonized column name (DATA-27) 

1237 "subject_id", 

1238 "participant", # also SMI BeGaze's event export 

1239 "participant_name", # Tobii Pro Lab `Participant name` / Tobii Studio `ParticipantName` 

1240 "recording_session_label", 

1241 "reader_id", 

1242 "USER", # Gazepoint 

1243 "recording_name", # Tobii Pro Lab — one recording per participant session 

1244 "recording_id", # Pupil Labs Neon 

1245] 

1246TRIAL_CANDIDATES = [ 

1247 "unique_trial_id", 

1248 "trial_id", 

1249 "unique_paragraph_id", 

1250 "paragraph_id", 

1251 "text_id", 

1252 "trial", 

1253 "trial_index", 

1254 "trial_number", # SMI BeGaze event export `Trial Number` 

1255 # One stimulus per trial is how Tobii, SMI and Gazepoint exports are 

1256 # shaped, so the stimulus name is the trial key when nothing above exists. 

1257 "presented_stimulus_name", # Tobii Pro Lab 

1258 "media_name", # Tobii Studio `MediaName` / Gazepoint `MEDIA_NAME` 

1259 "stimulus", # SMI BeGaze 

1260] 

1261SCREEN_ID_CANDIDATES = ["screen_id", "part_id", "page_id", "screen", "page", "part"] 

1262SCREEN_INDEX_CANDIDATES = [ 

1263 "screen_index", 

1264 "part_index", 

1265 "page_index", 

1266 "screen_order", 

1267 "page_number", 

1268] 

1269SCREEN_TIMESTAMP_CANDIDATES = [ 

1270 "screen_timestamp_ms", 

1271 "screen_time_ms", 

1272 "page_timestamp_ms", 

1273 "local_timestamp_ms", 

1274] 

1275SCREEN_FIXATION_ID_CANDIDATES = [ 

1276 "screen_fixation_id", 

1277 "screen_fixation_index", 

1278 "page_fixation_id", 

1279 "local_fixation_id", 

1280] 

1281CANVAS_WIDTH_CANDIDATES = ["canvas_width", "screen_width", "monitor_width"] 

1282CANVAS_HEIGHT_CANDIDATES = ["canvas_height", "screen_height", "monitor_height"] 

1283# Source column names that identify which *text* (passage) a row belongs to. 

1284# Output canonical column is `text_id` (was `paragraph_id`); the source names stay 

1285# as the real-world conventions so auto-detection keeps working. 

1286TEXT_ID_CANDIDATES = [ 

1287 "unique_paragraph_id", 

1288 "paragraph_id", 

1289 "unique_text_id", 

1290 "text_id", 

1291 "presented_stimulus_name", # Tobii Pro Lab — the stimulus *is* the text 

1292 "media_name", # Tobii Studio / Gazepoint 

1293 "stimulus", # SMI BeGaze 

1294 # #374 F13: an EyeLink export's own item column (`item`, `ITEM_ID`, …) — 

1295 # last, so a named text/paragraph column still wins. 

1296 "item", 

1297 "item_id", 

1298] 

1299TEXT_CANDIDATES = [ 

1300 "text", 

1301 "IA_LABEL", 

1302 "label", 

1303 "word", 

1304 "content", 

1305 "token", 

1306] 

1307# `word_idx` / `char_idx` are MultiplEYE's word- and character-level indices 

1308# (word_idx first so word-level boxes win over per-character ones). 

1309WORD_ID_CANDIDATES = [ 

1310 "word_id", 

1311 "IA_ID", 

1312 "ia_index", 

1313 "word_index", 

1314 "aoi", 

1315 "word_idx", 

1316 "char_idx", 

1317] 

1318LINE_CANDIDATES = ["line_idx", "line", "line_index", "IA_LINE_ID"] 

1319 

1320# `top_left_x` / `top_left_y` are MultiplEYE's box origin (paired with width/height). 

1321WORD_X_CANDIDATES = ["x", "left", "top_left_x"] 

1322WORD_Y_CANDIDATES = ["y", "top", "top_left_y"] 

1323WORD_WIDTH_CANDIDATES = ["width"] 

1324WORD_HEIGHT_CANDIDATES = ["height"] 

1325WORD_LEFT_CANDIDATES = ["IA_LEFT", "left", "start_x", "top_left_x"] 

1326WORD_RIGHT_CANDIDATES = ["IA_RIGHT", "right", "end_x"] 

1327WORD_TOP_CANDIDATES = ["IA_TOP", "top", "start_y", "top_left_y"] 

1328WORD_BOTTOM_CANDIDATES = ["IA_BOTTOM", "bottom", "end_y"] 

1329 

1330# `location_x` / `location_y` are MultiplEYE's fixation pixel coordinates. 

1331# Tobii Pro Lab: `Fixation point X [DACS px]`; Tobii Studio: `FixationPointX 

1332# (MCSpx)`; SMI BeGaze: `Position X [px]` / `Fixation Position X`; Pupil Labs 

1333# Neon: `fixation x [px]`. Gazepoint's FPOGX and Pupil Core's `norm_pos_x` are 

1334# screen *fractions* (0–1), not pixels — matched here so the column is found, 

1335# and reported as fractions by `screen_fraction_issues` (DATA-40), which is as 

1336# far as the load can go without knowing the screen size. 

1337FIX_X_CANDIDATES = [ 

1338 "x", 

1339 "CURRENT_FIX_X", 

1340 "FPOGX", 

1341 "location_x", 

1342 "fixation_point_x", 

1343 "fixation_x", 

1344 "position_x", 

1345 "fixation_position_x", 

1346 "norm_pos_x", 

1347] 

1348FIX_Y_CANDIDATES = [ 

1349 "y", 

1350 "CURRENT_FIX_Y", 

1351 "FPOGY", 

1352 "location_y", 

1353 "fixation_point_y", 

1354 "fixation_y", 

1355 "position_y", 

1356 "fixation_position_y", 

1357 "norm_pos_y", 

1358] 

1359FIX_DURATION_CANDIDATES = [ 

1360 "duration_ms", 

1361 "CURRENT_FIX_DURATION", 

1362 "CURRENT_FIX_LEN", 

1363 "duration", # also SMI `Duration [ms]`, Pupil Labs `duration [ms]` / `duration` 

1364 "fixation_duration", # SMI BeGaze `Fixation Duration [ms]` 

1365 "fix_duration", # EyeGenBench's own harmonized column name (DATA-27) 

1366 "eye_movement_event_duration", # Tobii Pro Lab (current name) 

1367 "gaze_event_duration", # Tobii Pro Lab (older) / Tobii Studio `GazeEventDuration` 

1368 "FPOGD", # Gazepoint — seconds, not ms (`time_unit_ms` converts, DATA-40) 

1369] 

1370FIX_TIMESTAMP_CANDIDATES = [ 

1371 "timestamp_ms", 

1372 "CURRENT_FIX_START", 

1373 "CURRENT_FIX_START_TIME", 

1374 "CURRENT_FIX_TIME", 

1375 "CURRENT_FIX_ONSET", 

1376 "onset", # MultiplEYE fixation onset (ms) 

1377 "fixation_start", # SMI BeGaze `Fixation Start [ms]` 

1378 "event_start_trial_time", # SMI BeGaze `Event Start Trial Time [ms]` 

1379 "start_timestamp", # Pupil Labs Core `start_timestamp` / Neon `start timestamp [ns]` 

1380 "recording_timestamp", # Tobii Pro Lab `Recording timestamp [ms]` / Studio `RecordingTimestamp` 

1381 "FPOGS", # Gazepoint — seconds 

1382] 

1383FIX_FIXATION_ID_CANDIDATES = [ 

1384 "fixation_id", # also Pupil Labs Neon `fixation id` 

1385 "CURRENT_FIX_INDEX", 

1386 "CURRENT_FIX_NUM", 

1387 "fixation_index", # also Tobii Studio `FixationIndex` 

1388 "fix_index", # EyeGenBench's own harmonized column name (DATA-27) 

1389 "eye_movement_type_index", # Tobii Pro Lab 

1390 "FPOGID", # Gazepoint 

1391] 

1392FIX_WORD_ID_CANDIDATES = [ 

1393 "word_id", 

1394 "IA_ID", 

1395 "ia_index", # EyeGenBench's own harmonized column name (DATA-27) 

1396 "CURRENT_FIX_INTEREST_AREA_ID", 

1397 "CURRENT_FIX_INTEREST_AREA_INDEX", 

1398 "word_index_in_text", 

1399 "word_index", 

1400 "word_idx", # MultiplEYE word index (resets per page) 

1401 "char_idx", # MultiplEYE character index 

1402] 

1403# Tobii Pro Lab `Gaze point X [DACS px]`, Tobii Studio `GazePointX`, Pupil Labs 

1404# Neon `gaze x [px]`, SMI raw IDF `L POR X [px]` (left eye first; the right eye 

1405# is the fallback when only it was recorded). 

1406RAW_GAZE_X_CANDIDATES = ["x", "FPOGX", "gaze_x", "gaze_point_x", "l_por_x", "r_por_x"] 

1407RAW_GAZE_Y_CANDIDATES = ["y", "FPOGY", "gaze_y", "gaze_point_y", "l_por_y", "r_por_y"] 

1408RAW_GAZE_TIMESTAMP_CANDIDATES = [ 

1409 "timestamp", # also Pupil Labs Neon `timestamp [ns]` 

1410 "time", # also SMI raw IDF `Time`, Gazepoint `TIME` 

1411 "ms", 

1412 "timestamp_ms", 

1413 "time_ms", 

1414 "recording_timestamp", # Tobii Pro Lab / Tobii Studio 

1415] 

1416 

1417 

1418_BOX_EDGES = ("left", "right", "top", "bottom") 

1419 

1420 

1421def _pick_box_edge_set(words: pd.DataFrame) -> dict[str, str] | None: 

1422 """The four word-box edge columns, resolved as one set (DATA-57). 

1423 

1424 ``pick_column`` looks at each edge on its own, and its second pass accepts a 

1425 prefixed or suffixed name (``LEFT_px``, ``aoi_left``) only when it is the 

1426 *only* column carrying that token. An AOI export routinely carries two box 

1427 encodings side by side — EyeLink's ``LEFT_px`` … ``BOTTOM_px`` next to a 

1428 derived ``aoi_left`` … ``aoi_bottom`` — so every edge was ambiguous and the 

1429 whole box landed in the manual step. The edges are not independent: they 

1430 share an affix. So each column naming exactly one edge is keyed by the rest 

1431 of its name (``*_px``, ``aoi_*``), and a key that covers all four edges is a 

1432 set. The set that comes **last** in the table wins: a derived box is 

1433 usually appended after the one the export shipped with, and the columns a 

1434 lab adds later are the ones it means (``aoi_left`` … then ``LEFT_px`` … 

1435 picks ``LEFT_px``). 

1436 

1437 Returns ``{edge: column}`` plus the shared affix under ``"affix"`` (for 

1438 ``_affix_sibling``), or ``None`` when no complete set exists.""" 

1439 groups: dict[tuple[str, ...], dict[str, str]] = {} 

1440 order: dict[tuple[str, ...], int] = {} 

1441 for pos, col in enumerate(words.columns): 

1442 tokens = _col_tokens(col) 

1443 edges = [tok for tok in tokens if tok in _BOX_EDGES] 

1444 if len(edges) != 1: 

1445 continue 

1446 affix = tuple("*" if tok == edges[0] else tok for tok in tokens) 

1447 group = groups.setdefault(affix, {}) 

1448 if edges[0] not in group: 

1449 group[edges[0]] = col 

1450 order.setdefault(affix, pos) 

1451 complete = [a for a, g in groups.items() if len(g) == len(_BOX_EDGES)] 

1452 if not complete: 

1453 return None 

1454 affix = max(complete, key=order.__getitem__) 

1455 return {**groups[affix], "affix": affix} 

1456 

1457 

1458def _affix_sibling( 

1459 words: pd.DataFrame, affix: tuple[str, ...], token: str 

1460) -> str | None: 

1461 """The column named like an edge set's affix with ``token`` in the edge's 

1462 place — ``aoi_width`` beside ``aoi_left`` … ``aoi_bottom``.""" 

1463 want = [token if tok == "*" else tok for tok in affix] 

1464 return next((col for col in words.columns if _col_tokens(col) == want), None) 

1465 

1466 

1467#: AN-32 — the per-AOI reading measures a dataset *brings*: the Corpus Analysis 

1468#: page shows these and computes none of them. Each is an optional AOI-table 

1469#: field, ``(schema key, canonical column, short label, full name, kind, 

1470#: candidates)``; the candidates lead with EyeLink Data Viewer's interest-area 

1471#: report names (``IA_*``), then the plain spellings other exports use. The 

1472#: canonical columns are ``aggregation.MEASURES``' own, so a mapped measure is 

1473#: the one the page's pickers offer. 

1474READING_MEASURE_FIELDS: tuple[tuple[str, str, str, str, str, tuple[str, ...]], ...] = ( 

1475 ( 

1476 "measure_tfd", 

1477 "total_fixation_duration_ms", 

1478 "TFD", 

1479 "Total fixation duration (dwell time), ms", 

1480 "numeric", 

1481 ("IA_DWELL_TIME", "total_fixation_duration", "TFD", "dwell_time"), 

1482 ), 

1483 ( 

1484 "measure_ffd", 

1485 "first_fixation_ms", 

1486 "FFD", 

1487 "First fixation duration, ms", 

1488 "numeric", 

1489 ("IA_FIRST_FIXATION_DURATION", "first_fixation_duration", "FFD"), 

1490 ), 

1491 ( 

1492 "measure_fprt", 

1493 "first_pass_gaze_duration_ms", 

1494 "FPRT", 

1495 "First-pass reading time (gaze duration), ms", 

1496 "numeric", 

1497 ( 

1498 "IA_FIRST_RUN_DWELL_TIME", 

1499 "first_pass_gaze_duration", 

1500 "gaze_duration", 

1501 "FPRT", 

1502 ), 

1503 ), 

1504 ( 

1505 "measure_rpd", 

1506 "regression_path_duration_ms", 

1507 "RPD", 

1508 "Regression-path (go-past) duration, ms", 

1509 "numeric", 

1510 ("IA_REGRESSION_PATH_DURATION", "regression_path_duration", "go_past", "RPD"), 

1511 ), 

1512 ( 

1513 "measure_second_pass", 

1514 "second_pass_duration_ms", 

1515 "2nd pass", 

1516 "Second-pass duration, ms", 

1517 "numeric", 

1518 ("IA_SECOND_RUN_DWELL_TIME", "second_pass_duration"), 

1519 ), 

1520 ( 

1521 "measure_single_fix", 

1522 "single_fixation_duration_ms", 

1523 "Single fix.", 

1524 "Single-fixation duration, ms", 

1525 "numeric", 

1526 ("IA_SINGLE_FIXATION_DURATION", "single_fixation_duration", "SFD"), 

1527 ), 

1528 ( 

1529 "measure_nfix", 

1530 "n_fixations", 

1531 "Fix. count", 

1532 "Number of fixations on the word", 

1533 "numeric", 

1534 ("IA_FIXATION_COUNT", "fixation_count", "n_fixations"), 

1535 ), 

1536 ( 

1537 "measure_skip", 

1538 "skip_flag", 

1539 "Skip", 

1540 "Skipped in first pass (0/1)", 

1541 "boolean", 

1542 ("IA_SKIP", "skip_flag", "skipped"), 

1543 ), 

1544 ( 

1545 "measure_reg_in", 

1546 "regression_in_flag", 

1547 "Reg. in", 

1548 "Regressed into (0/1)", 

1549 "boolean", 

1550 ("IA_REGRESSION_IN", "regression_in_flag"), 

1551 ), 

1552 ( 

1553 "measure_reg_out", 

1554 "regression_out_flag", 

1555 "Reg. out", 

1556 "Regressed out of (0/1)", 

1557 "boolean", 

1558 ("IA_REGRESSION_OUT", "regression_out_flag"), 

1559 ), 

1560 ( 

1561 "measure_reg_in_count", 

1562 "number_of_regressions_in", 

1563 "Reg. in (n)", 

1564 "Number of regressions into the word", 

1565 "numeric", 

1566 ("IA_REGRESSION_IN_COUNT", "number_of_regressions_in", "regression_in_count"), 

1567 ), 

1568 ( 

1569 "measure_landing_position", 

1570 "initial_landing_position", 

1571 "Landing pos.", 

1572 "Initial landing position, letters", 

1573 "numeric", 

1574 ("IA_FIRST_FIXATION_LANDING_POSITION", "initial_landing_position"), 

1575 ), 

1576 ( 

1577 "measure_landing_distance", 

1578 "initial_landing_distance", 

1579 "Landing dist.", 

1580 "Centered initial landing distance, letters", 

1581 "numeric", 

1582 ("initial_landing_distance", "landing_distance"), 

1583 ), 

1584) 

1585READING_MEASURE_KEYS: tuple[str, ...] = tuple(f[0] for f in READING_MEASURE_FIELDS) 

1586READING_MEASURE_COLUMNS: tuple[str, ...] = tuple(f[1] for f in READING_MEASURE_FIELDS) 

1587 

1588 

1589def brought_reading_measures(words: pd.DataFrame | None) -> list[str]: 

1590 """The reading-measure columns ``words`` carries — what the dataset brought. 

1591 

1592 AN-32 / EXP-23: the Corpus Analysis page and the export show these and 

1593 compute none, so this only *reads* the frame; an empty list means the 

1594 dataset has no reading measures.""" 

1595 if words is None: 

1596 return [] 

1597 return [column for column in READING_MEASURE_COLUMNS if column in words.columns] 

1598 

1599 

1600def _apply_reading_measures( 

1601 df: pd.DataFrame, source: pd.DataFrame, schema: dict 

1602) -> None: 

1603 """Write the mapped reading measures onto ``df`` (AN-32). 

1604 

1605 A schema that names a measure key decides that measure outright: mapped, the 

1606 column is copied under its canonical name (and wins over the optional-field 

1607 passthrough); cleared, the canonical column is removed even if the 

1608 passthrough carried it — "this dataset has no TFD" has to stay true. A 

1609 schema without measure keys (a dataset stored before AN-32) is left as the 

1610 passthrough made it.""" 

1611 for key, canonical, _label, _name, kind, _candidates in READING_MEASURE_FIELDS: 

1612 if key not in schema: 

1613 continue 

1614 column = schema.get(key) 

1615 if column and column in source.columns: 

1616 values = source[column] 

1617 df[canonical] = ( 

1618 coerce_measure_flag(values) if kind == "boolean" else _to_number(values) 

1619 ) 

1620 elif canonical in df.columns: 

1621 del df[canonical] 

1622 

1623 

1624#: The duration measures a word nobody fixated does not have (BUG-63). Total 

1625#: fixation duration is not one of them: the word was read past and got 0 ms. 

1626UNFIXATED_BLANK_MEASURES: tuple[str, ...] = ( 

1627 "first_fixation_ms", 

1628 "first_pass_gaze_duration_ms", 

1629 "regression_path_duration_ms", 

1630 "single_fixation_duration_ms", 

1631) 

1632 

1633 

1634def _blank_unfixated_measures(df: pd.DataFrame) -> None: 

1635 """Blank the imported FFD / FPRT / RPD / single-fixation duration of a word 

1636 nobody fixated (BUG-63's rule, on the imported path — #374 F1). 

1637 

1638 An EyeLink IA report writes ``0`` there, and every mean then counted a 

1639 skipped word as a 0 ms fixation. "Never fixated" is a fixation count of 0; 

1640 where no count is mapped (or the cell is blank), a total fixation duration 

1641 of 0 says the same. Total fixation duration itself keeps its 0. 

1642 

1643 The mirror image, as on the computed path: second pass is "fewer than two 

1644 runs ⇒ 0", but an imported IA_SECOND_RUN_DWELL_TIME leaves those cells 

1645 blank, so its mean would cover only re-read words. Where the fixation count 

1646 is known, a blank second pass becomes 0.""" 

1647 if "second_pass_duration_ms" in df.columns and "n_fixations" in df.columns: 

1648 known = pd.to_numeric(df["n_fixations"], errors="coerce").notna() 

1649 second = pd.to_numeric(df["second_pass_duration_ms"], errors="coerce") 

1650 df["second_pass_duration_ms"] = second.mask(known & second.isna(), 0.0) 

1651 present = [column for column in UNFIXATED_BLANK_MEASURES if column in df.columns] 

1652 if not present: 

1653 return 

1654 unfixated = pd.Series(False, index=df.index) 

1655 count = ( 

1656 pd.to_numeric(df["n_fixations"], errors="coerce") 

1657 if "n_fixations" in df.columns 

1658 else pd.Series(np.nan, index=df.index) 

1659 ) 

1660 unfixated |= count.eq(0) 

1661 if "total_fixation_duration_ms" in df.columns: 

1662 total = pd.to_numeric(df["total_fixation_duration_ms"], errors="coerce") 

1663 unfixated |= count.isna() & total.eq(0) 

1664 if not unfixated.any(): 

1665 return 

1666 for column in present: 

1667 df[column] = pd.to_numeric(df[column], errors="coerce").mask(unfixated) 

1668 

1669 

1670def propose_word_schema(words: pd.DataFrame) -> dict[str, str | None]: 

1671 """Return a candidate column mapping for words/IA data without erroring.""" 

1672 schema = _propose_word_schema_by_field(words) 

1673 # BUG-99: a box field is a number, so a column with no number in it is not 

1674 # one, whatever its name says. OneStop's IA report carries `TOP_LEFT`, the 

1675 # box corner as the text `(368,186)`; its tokens matched both the x and the 

1676 # y list, so the load proposed it for both and warned about every cell. 

1677 for key in _BOX_FIELDS: 

1678 if schema[key] and _holds_no_number(words[schema[key]]): 

1679 schema[key] = None 

1680 # AN-32: every reading measure is proposed too, from its known names. 

1681 # The app's own canonical name is a candidate too, right after EyeLink's, 

1682 # so a table the app itself wrote (an exported words.csv, a normalized 

1683 # frame handed to the API) keeps its measures instead of having the key 

1684 # proposed empty — which `_apply_reading_measures` reads as "absent". 

1685 for key, column, _label, _name, _kind, candidates in READING_MEASURE_FIELDS: 

1686 schema[key] = pick_column(words, (candidates[0], column, *candidates[1:])) 

1687 if all(schema[edge] for edge in _BOX_EDGES): 

1688 return schema 

1689 edge_set = _pick_box_edge_set(words) 

1690 if edge_set is None: 

1691 return schema 

1692 affix = edge_set.pop("affix") 

1693 schema.update(edge_set) 

1694 # The origin + size fields follow the same set, so the two encodings the 

1695 # mapping screen offers describe one box rather than two. 

1696 schema["x"] = schema["x"] or edge_set["left"] 

1697 schema["y"] = schema["y"] or edge_set["top"] 

1698 for size in ("width", "height"): 

1699 schema[size] = _affix_sibling(words, affix, size) 

1700 return schema 

1701 

1702 

1703_BOX_FIELDS = ("x", "y", "width", "height", *_BOX_EDGES) 

1704 

1705 

1706def _holds_no_number(values: pd.Series) -> bool: 

1707 """Whether ``values`` has filled cells and none of them reads as a number. 

1708 

1709 A header-only frame (the mapping screen proposes from one) has no cells and 

1710 so is never ruled out; a numeric column with a few bad cells is still a 

1711 numeric column, and stays proposed for :func:`numeric_parse_issues` to 

1712 report.""" 

1713 if pd.api.types.is_numeric_dtype(values) or pd.api.types.is_bool_dtype(values): 

1714 return False 

1715 filled = _filled_cells(values) 

1716 return not filled.empty and _to_number(filled).isna().all() 

1717 

1718 

1719def _propose_word_schema_by_field(words: pd.DataFrame) -> dict[str, str | None]: 

1720 return dict( 

1721 participant=pick_column(words, PARTICIPANT_CANDIDATES), 

1722 trial=pick_column(words, TRIAL_CANDIDATES), 

1723 screen_id=pick_column(words, SCREEN_ID_CANDIDATES), 

1724 screen_index=pick_column(words, SCREEN_INDEX_CANDIDATES), 

1725 canvas_width=pick_column(words, CANVAS_WIDTH_CANDIDATES), 

1726 canvas_height=pick_column(words, CANVAS_HEIGHT_CANDIDATES), 

1727 text_id=pick_column(words, TEXT_ID_CANDIDATES), 

1728 word_id=pick_column(words, WORD_ID_CANDIDATES), 

1729 text=pick_column(words, TEXT_CANDIDATES), 

1730 line=pick_column(words, LINE_CANDIDATES), 

1731 x=pick_column(words, WORD_X_CANDIDATES), 

1732 y=pick_column(words, WORD_Y_CANDIDATES), 

1733 width=pick_column(words, WORD_WIDTH_CANDIDATES), 

1734 height=pick_column(words, WORD_HEIGHT_CANDIDATES), 

1735 left=pick_column(words, WORD_LEFT_CANDIDATES), 

1736 right=pick_column(words, WORD_RIGHT_CANDIDATES), 

1737 top=pick_column(words, WORD_TOP_CANDIDATES), 

1738 bottom=pick_column(words, WORD_BOTTOM_CANDIDATES), 

1739 ) 

1740 

1741 

1742def propose_fix_schema(fixations: pd.DataFrame) -> dict[str, str | None]: 

1743 """Return a candidate column mapping for fixations data without erroring. 

1744 

1745 pass_index / saccade_type / saccade_amplitude / eye are not schema fields — 

1746 they're auto-detected and kept via ``FIX_OPTIONAL_FIELDS`` (and offered under 

1747 *fields to keep*), so they're not proposed here.""" 

1748 return dict( 

1749 participant=pick_column(fixations, PARTICIPANT_CANDIDATES), 

1750 trial=pick_column(fixations, TRIAL_CANDIDATES), 

1751 screen_id=pick_column(fixations, SCREEN_ID_CANDIDATES), 

1752 screen_index=pick_column(fixations, SCREEN_INDEX_CANDIDATES), 

1753 screen_timestamp=pick_column(fixations, SCREEN_TIMESTAMP_CANDIDATES), 

1754 screen_fixation_id=pick_column(fixations, SCREEN_FIXATION_ID_CANDIDATES), 

1755 canvas_width=pick_column(fixations, CANVAS_WIDTH_CANDIDATES), 

1756 canvas_height=pick_column(fixations, CANVAS_HEIGHT_CANDIDATES), 

1757 text_id=pick_column(fixations, TEXT_ID_CANDIDATES), 

1758 fixation_id=pick_column(fixations, FIX_FIXATION_ID_CANDIDATES), 

1759 timestamp=pick_column(fixations, FIX_TIMESTAMP_CANDIDATES), 

1760 duration=pick_column(fixations, FIX_DURATION_CANDIDATES), 

1761 x=pick_column(fixations, FIX_X_CANDIDATES), 

1762 y=pick_column(fixations, FIX_Y_CANDIDATES), 

1763 word_id=pick_column(fixations, FIX_WORD_ID_CANDIDATES), 

1764 ) 

1765 

1766 

1767def propose_raw_gaze_schema(raw_gaze: pd.DataFrame) -> dict[str, str | None]: 

1768 """Return a candidate column mapping for raw gaze data without erroring.""" 

1769 return dict( 

1770 participant=pick_column(raw_gaze, PARTICIPANT_CANDIDATES), 

1771 trial=pick_column(raw_gaze, TRIAL_CANDIDATES), 

1772 screen_id=pick_column(raw_gaze, SCREEN_ID_CANDIDATES), 

1773 screen_index=pick_column(raw_gaze, SCREEN_INDEX_CANDIDATES), 

1774 text_id=pick_column(raw_gaze, TEXT_ID_CANDIDATES), 

1775 word_id=pick_column(raw_gaze, FIX_WORD_ID_CANDIDATES), 

1776 text=pick_column(raw_gaze, TEXT_CANDIDATES), 

1777 x=pick_column(raw_gaze, RAW_GAZE_X_CANDIDATES), 

1778 y=pick_column(raw_gaze, RAW_GAZE_Y_CANDIDATES), 

1779 timestamp=pick_column(raw_gaze, RAW_GAZE_TIMESTAMP_CANDIDATES), 

1780 ) 

1781 

1782 

1783def validate_word_schema(schema: dict[str, str | None]) -> list: 

1784 """Return a list of human-readable problems with a words/IA schema. 

1785 

1786 Participant ID is optional: word/AoI tables without one are treated as 

1787 stimulus-level (one row per word per *text*, not per reading) and are 

1788 broadcast across the participants found in the fixations — see 

1789 ``broadcast_stimulus_words``.""" 

1790 problems = [] 

1791 for key, label in [ 

1792 ("trial", "Trial ID"), 

1793 ("word_id", "Word/IA ID"), 

1794 ]: 

1795 if not schema.get(key): 

1796 problems.append(f"missing {label}") 

1797 has_xywh = all(schema.get(k) for k in ["x", "y", "width", "height"]) 

1798 has_box = all(schema.get(k) for k in ["left", "right", "top", "bottom"]) 

1799 if not has_xywh and not has_box: 

1800 problems.append("missing Word box (its edges, or x/y with width and height)") 

1801 return problems 

1802 

1803 

1804def validate_fix_schema(schema: dict[str, str | None]) -> list: 

1805 """Return a list of human-readable problems with a fixations schema. 

1806 

1807 X/Y coordinates are optional when a Word/IA ID is mapped: AOI-sequence 

1808 datasets (fixations recorded as "which word", not "which pixel") get 

1809 coordinates from the matching word-box centers — see 

1810 ``fill_fixation_xy_from_words``.""" 

1811 problems = [] 

1812 # Participant is optional — a dataset without it is treated as a single 

1813 # anonymous reader (see SYNTHETIC_PARTICIPANT). 

1814 for key, label in [ 

1815 ("trial", "Trial ID"), 

1816 ("duration", "Duration"), 

1817 ]: 

1818 if not schema.get(key): 

1819 problems.append(f"missing {label}") 

1820 has_xy = schema.get("x") and schema.get("y") 

1821 if not has_xy and not schema.get("word_id"): 

1822 problems.append( 

1823 "missing X and Y (or a Word/IA ID, to place each fixation at its " 

1824 "word's center)" 

1825 ) 

1826 return problems 

1827 

1828 

1829def validate_raw_gaze_schema(schema: dict[str, str | None]) -> list: 

1830 """Return a list of human-readable problems with a raw gaze schema.""" 

1831 problems = [] 

1832 # Participant optional — single anonymous reader when absent. 

1833 for key, label in [ 

1834 ("trial", "Trial ID"), 

1835 ("x", "X"), 

1836 ("y", "Y"), 

1837 ]: 

1838 if not schema.get(key): 

1839 problems.append(f"missing {label}") 

1840 return problems 

1841 

1842 

1843# Column added when concatenating several files (multi-file upload, glob, or a 

1844# multi-member zip): the source file's stem. Lets datasets that key metadata in 

1845# the *filename* (one file per participant and/or per text, e.g. PoTeC's 

1846# `reader0_b0_scanpath.tsv`) recover it after concatenation — map it as (part 

1847# of) the Trial/Participant ID. 

1848SOURCE_FILE_COLUMN = "source_file" 

1849 

1850# Prefix for the positional columns split out of `source_file` by 

1851# `split_source_file` (file_part_1, file_part_2, …). 

1852FILE_PART_PREFIX = "file_part_" 

1853 

1854TablesInput = str | os.PathLike | object | list 

1855 

1856 

1857def split_source_file( 

1858 df: pd.DataFrame, 

1859 *, 

1860 delimiter: str = "_", 

1861 column: str = SOURCE_FILE_COLUMN, 

1862 prefix: str = FILE_PART_PREFIX, 

1863) -> pd.DataFrame: 

1864 """Split a ``source_file`` column into positional ``file_part_N`` columns. 

1865 

1866 Lets the upload wizard derive a trial / participant id from a structured 

1867 filename when no data column carries it — e.g. ``reader0_b0_scanpath`` split 

1868 on ``_`` yields ``file_part_1=reader0``, ``file_part_2=b0``, 

1869 ``file_part_3=scanpath`` (the user then maps the relevant part(s), composing 

1870 several if needed). Returns ``df`` unchanged if ``column`` is absent or 

1871 ``delimiter`` is empty. Rows with fewer parts get empty strings for the 

1872 missing tail, so every row has the same part columns.""" 

1873 if column not in df.columns or not delimiter: 

1874 return df 

1875 parts = ( 

1876 df[column].astype(str).str.split(delimiter, expand=True, regex=False).fillna("") 

1877 ) 

1878 df = df.copy() 

1879 for i in range(parts.shape[1]): 

1880 df[f"{prefix}{i + 1}"] = parts[i].to_numpy() 

1881 return df 

1882 

1883 

1884def extract_columns_from_source_file( 

1885 df: pd.DataFrame, 

1886 pattern: str, 

1887 *, 

1888 column: str = SOURCE_FILE_COLUMN, 

1889 lowercase: bool = False, 

1890) -> pd.DataFrame: 

1891 """Add one column per *named group* of a regex applied to ``source_file``. 

1892 

1893 Sibling of :func:`split_source_file` for filenames whose fields are 

1894 positionally irregular — varying-length parts or an optional prefix make a 

1895 fixed delimiter split unreliable. A regex with named groups, e.g. 

1896 ``r"(?P<session>\\d+_\\w+_ET\\d)_.*_(?P<stimulus>.+)_scanpath"``, extracts each 

1897 group into its own column the wizard can then map as a trial / participant id. 

1898 ``lowercase`` folds the captured values (useful when one table names a field 

1899 CamelCase and another lowercase). No-op (returns ``df`` unchanged) when 

1900 ``column`` is absent, ``pattern`` is empty / uncompilable, or it declares no 

1901 named groups; rows that don't match get NaN. A named group that collides with 

1902 an existing column is **skipped** (the real data wins) — see 

1903 :func:`source_file_regex_collisions` to surface those in a UI.""" 

1904 if not pattern or column not in df.columns: 

1905 return df 

1906 try: 

1907 compiled = re.compile(pattern) 

1908 except re.error: 

1909 return df 

1910 if not compiled.groupindex: 

1911 return df 

1912 extracted = df[column].astype(str).str.extract(compiled) 

1913 df = df.copy() 

1914 for group in compiled.groupindex: # named groups only 

1915 if group in df.columns: # don't clobber an existing data column 

1916 continue 

1917 values = extracted[group] 

1918 if lowercase: 

1919 values = values.str.lower() 

1920 df[group] = values.to_numpy() 

1921 return df 

1922 

1923 

1924def source_file_regex_collisions( 

1925 df: pd.DataFrame, pattern: str, *, column: str = SOURCE_FILE_COLUMN 

1926) -> list: 

1927 """Named groups of ``pattern`` that already exist as columns in ``df``. 

1928 

1929 :func:`extract_columns_from_source_file` skips these (so it never clobbers 

1930 real data); the wizard surfaces them so the user can rename the group. 

1931 ``column`` is accepted for symmetry with its two siblings (UX-113 — 

1932 "derive from any column", not only ``source_file``) but isn't otherwise 

1933 used: a collision is a collision against ``df``'s columns regardless of 

1934 which column the regex reads from.""" 

1935 if not pattern: 

1936 return [] 

1937 try: 

1938 groups = re.compile(pattern).groupindex 

1939 except re.error: 

1940 return [] 

1941 return [g for g in groups if g in df.columns] 

1942 

1943 

1944def _renumber_word_ids_across_blocks( 

1945 out: pd.DataFrame, *, screen_cols: list, block_col: str, word_col: str 

1946) -> pd.DataFrame: 

1947 """Reassign ``word_col`` sequentially per screen, in block-then-original 

1948 order — the post-aggregation half of :func:`aggregate_char_boxes`'s block 

1949 support (see its docstring). ``out`` is the already-aggregated frame (one 

1950 row per word box), still carrying the ``_box_l``/``_box_t`` temp columns 

1951 :func:`aggregate_char_boxes` computes just before calling this. A block's 

1952 reading-order position is its own first box's top-then-left corner — 

1953 geometry the aggregation already has, not a second field to ask for.""" 

1954 block_order = ( 

1955 out.groupby([*screen_cols, block_col], sort=False)[["_box_t", "_box_l"]] 

1956 .first() 

1957 .reset_index() 

1958 .sort_values([*screen_cols, "_box_t", "_box_l"]) 

1959 ) 

1960 block_order["_block_rank"] = block_order.groupby(screen_cols, sort=False).cumcount() 

1961 out = out.merge( 

1962 block_order[[*screen_cols, block_col, "_block_rank"]], 

1963 on=[*screen_cols, block_col], 

1964 how="left", 

1965 ) 

1966 out = out.sort_values([*screen_cols, "_block_rank", word_col], kind="stable") 

1967 out[word_col] = out.groupby(screen_cols, sort=False).cumcount() 

1968 return out.drop(columns="_block_rank").reset_index(drop=True) 

1969 

1970 

1971def aggregate_char_boxes( 

1972 df: pd.DataFrame, schema: dict[str, str | None] 

1973) -> pd.DataFrame: 

1974 """Collapse character-level AOI rows into one bounding box per word. 

1975 

1976 For interest-area tables shipped one row per *character* (e.g. CJK corpora 

1977 that have no whitespace word boundaries), aggregate the characters of each 

1978 word — grouped by the mapped trial id + word id (plus participant / text id / 

1979 screen id when mapped) — into a single bounding box: min/max over the mapped 

1980 box columns, first value of every other column. A mapped screen id narrows 

1981 the group the same way participant/text id do — a multi-screen (multipart) 

1982 trial commonly restarts word numbering per screen, so without it a shared 

1983 word id on two screens of the same trial would silently merge into one box. 

1984 Run this on the RAW frame *before* 

1985 :func:`normalize_words` (which expects one row per word box). ``schema`` is a 

1986 word schema dict (field → source column). Returns ``df`` unchanged when the 

1987 trial or word-id column isn't mapped, or no box columns are. 

1988 

1989 UX-113 — ``schema["block"]``, when mapped, groups words into sub-screen 

1990 blocks that each restart their own local word numbering (e.g. a 

1991 comprehension question's stem/target/distractor answer blocks). Without 

1992 it, two blocks' word 0 would silently aggregate into one merged box. When 

1993 it *is* mapped, the word id is also renumbered sequentially per screen 

1994 afterwards — in block-reading-order (each block's own top-then-left 

1995 position, geometry rather than a second field the user would have to 

1996 supply) then original word-id order — so ids stay unique within one 

1997 screen, which matters because a mapped word id can be authoritative for 

1998 fixation assignment (see ``measures.assign_fixations_to_words``).""" 

1999 word_col = schema.get("word_id") 

2000 trial = schema.get("trial") 

2001 if not word_col or not trial: 

2002 return df 

2003 block_col = schema.get("block") or None 

2004 group_cols = list(trial_mapping_columns(trial)) 

2005 # A mapped screen id must narrow the group like participant/text_id do — 

2006 # a multi-screen trial (multipart) commonly restarts word numbering per 

2007 # screen (e.g. MultiplEYE's reading pages), so without this a shared 

2008 # word id on two screens of the same trial would silently merge into one 

2009 # box, the same failure mode "block" fixes within a single screen. 

2010 for key in ("participant", "text_id", "screen_id"): 

2011 mapped = schema.get(key) 

2012 if mapped: 

2013 group_cols += trial_mapping_columns(mapped) 

2014 if block_col: 

2015 group_cols.append(block_col) 

2016 group_cols.append(word_col) 

2017 # De-dup, keep only columns actually present, and require the word id. 

2018 group_cols = [c for c in dict.fromkeys(group_cols) if c in df.columns] 

2019 if word_col not in group_cols: 

2020 return df 

2021 if block_col not in group_cols: # mapped, but not actually on this frame 

2022 block_col = None 

2023 

2024 has_xywh = all(schema.get(k) for k in ("x", "y", "width", "height")) 

2025 has_edges = all(schema.get(k) for k in ("left", "right", "top", "bottom")) 

2026 if not has_xywh and not has_edges: 

2027 return df 

2028 

2029 df = df.copy() 

2030 if has_xywh: 

2031 left = _to_number(df[schema["x"]]) 

2032 top = _to_number(df[schema["y"]]) 

2033 df["_box_l"], df["_box_t"] = left, top 

2034 df["_box_r"] = left + _to_number(df[schema["width"]]) 

2035 df["_box_b"] = top + _to_number(df[schema["height"]]) 

2036 else: 

2037 df["_box_l"] = _to_number(df[schema["left"]]) 

2038 df["_box_r"] = _to_number(df[schema["right"]]) 

2039 df["_box_t"] = _to_number(df[schema["top"]]) 

2040 df["_box_b"] = _to_number(df[schema["bottom"]]) 

2041 

2042 temp = {"_box_l", "_box_r", "_box_t", "_box_b"} 

2043 agg = {c: "first" for c in df.columns if c not in group_cols and c not in temp} 

2044 agg.update(_box_l="min", _box_t="min", _box_r="max", _box_b="max") 

2045 out = df.groupby(group_cols, sort=False, as_index=False).agg(agg) 

2046 

2047 if block_col: 

2048 out = _renumber_word_ids_across_blocks( 

2049 out, 

2050 screen_cols=[c for c in group_cols if c not in (word_col, block_col)], 

2051 block_col=block_col, 

2052 word_col=word_col, 

2053 ) 

2054 

2055 # Write the aggregated box back into the SAME schema columns so the existing 

2056 # word schema still maps it (origin+size or edges, matching the input form). 

2057 if has_xywh: 

2058 out[schema["x"]] = out["_box_l"] 

2059 out[schema["y"]] = out["_box_t"] 

2060 out[schema["width"]] = out["_box_r"] - out["_box_l"] 

2061 out[schema["height"]] = out["_box_b"] - out["_box_t"] 

2062 else: 

2063 out[schema["left"]] = out["_box_l"] 

2064 out[schema["right"]] = out["_box_r"] 

2065 out[schema["top"]] = out["_box_t"] 

2066 out[schema["bottom"]] = out["_box_b"] 

2067 return out.drop(columns=list(temp)) 

2068 

2069 

2070def _read_by_extension( 

2071 buf, name: str, plan: ReadPlan | None = None, *, sep: str | None = None 

2072) -> pd.DataFrame: 

2073 """Dispatch a buffer/path to a pandas reader by its (lowercased) name. 

2074 

2075 ``sep`` is the delimiter of a text table when the caller already knows it 

2076 (a zip member, which cannot be peeked at); otherwise it is read off the 

2077 header line (DATA-41). 

2078 

2079 ``plan`` (PERF-6) narrows the read to the columns normalization keeps and 

2080 declares EyeLink's ``.`` missing in the numeric ones. Delimited text and 

2081 Parquet honour it; Excel has no column-projection reader, so it is read 

2082 whole and pruned after. 

2083 

2084 CSV/TSV reads pass ``low_memory=False`` so pandas infers one dtype per 

2085 column in a single pass. The default chunked parser can otherwise read the 

2086 same column as numeric in early chunks and as strings in a later chunk that 

2087 holds a sentinel (e.g. EyeLink's ``.`` in ``CURRENT_FIX_PRECISION_MEASURE_*`` 

2088 columns), leaving a single ``object`` column that mixes Python ``float`` and 

2089 ``str`` values. Such a column emits a ``DtypeWarning`` and later crashes 

2090 pyarrow when Streamlit serializes the frame for display — only on large 

2091 (multi-chunk) files, which is why a small upload reads fine locally but a 

2092 full report kills the worker on the cloud. Matches the other read paths 

2093 (``load_onestop_server_bundle``, ``onestop_shard``).""" 

2094 columns = list(plan.columns) if plan is not None and plan.columns else None 

2095 if name.endswith(".parquet"): 

2096 return pd.read_parquet(buf, columns=columns) 

2097 if name.endswith(".feather"): 

2098 return pd.read_feather(buf, columns=columns) 

2099 if name.endswith((".xlsx", ".xls")): 

2100 if not _is_workbook(buf, name): 

2101 # BUG-55: EyeLink Data Viewer's "Excel" export is tab-separated text 

2102 # with an .xls name — read it as what it is. 

2103 return _read_delimited(buf, sep or _sniff_delimiter(buf, name), plan) 

2104 # First sheet (e.g. MultiplEYE questions workbook). 

2105 frame = pd.read_excel(buf, **_excel_na_kwargs(buf, plan)) 

2106 return frame[[c for c in columns if c in frame.columns]] if columns else frame 

2107 return _read_delimited(buf, sep or _sniff_delimiter(buf, name), plan) 

2108 

2109 

2110#: The delimiters a text table is looked for with, and the one assumed when its 

2111#: header line settles nothing (DATA-41). 

2112_DELIMITERS = ("\t", ",", ";", "|") 

2113_QUOTED = re.compile(r'"[^"]*"') 

2114 

2115 

2116def _default_delimiter(name: str) -> str: 

2117 """The delimiter a text file's extension implies: tab for ``.tsv`` / 

2118 ``.tab`` and for the tab-separated exports named ``.txt`` or ``.xls``, 

2119 comma for everything else.""" 

2120 return "\t" if name.lower().endswith((".tsv", ".tab", ".txt", ".xls")) else "," 

2121 

2122 

2123def _delimiter_of(header_line: bytes, name: str) -> str: 

2124 """The delimiter a table's header line uses (DATA-41). 

2125 

2126 A ``;``-separated CSV (Excel's export wherever the decimal separator is a 

2127 comma) and a tab-separated ``.txt`` both used to load as a single column 

2128 holding the whole line. Counting each candidate in the header, outside 

2129 quotes, is enough: a column name never contains the delimiter, while a 

2130 sniffer that also reads the rows is misled by the decimal commas in them. 

2131 A ``.tsv`` is always tab-separated; a header with no candidate in it (a 

2132 one-column table) keeps the extension's default. 

2133 """ 

2134 default = _default_delimiter(name) 

2135 if name.lower().endswith((".tsv", ".tab")): 

2136 return default 

2137 text = _QUOTED.sub("", header_line.decode("latin-1")) 

2138 counts = {sep: text.count(sep) for sep in _DELIMITERS} 

2139 best = max(counts, key=lambda sep: counts[sep]) 

2140 return best if counts[best] > counts[default] else default 

2141 

2142 

2143def _first_line(head: bytes) -> bytes: 

2144 """The header line of a text table's opening bytes. 

2145 

2146 A UTF-16 or UTF-32 file (EyeLink Data Viewer's default export) is decoded 

2147 first, so the line is cut at its newline character rather than at the 

2148 first newline *byte*, and comes back as UTF-8. 

2149 """ 

2150 encoding = _bom_encoding(head) 

2151 if encoding is not None and encoding != "utf-8-sig": 

2152 line = head.decode(encoding, "ignore").split("\n", 1)[0].rstrip("\r") 

2153 return line.encode("utf-8") 

2154 return head.split(b"\n", 1)[0].rstrip(b"\r") 

2155 

2156 

2157def _sniff_delimiter(buf, name: str) -> str: 

2158 """:func:`_delimiter_of` for an upload or a path, read without consuming 

2159 it; a stream that cannot be rewound keeps the extension's default.""" 

2160 if not _can_reread(buf): 

2161 return _default_delimiter(name) 

2162 return _delimiter_of(_first_line(_peek(buf, _HEADER_MAX_BYTES)), name) 

2163 

2164 

2165#: The first bytes of a legacy (OLE2) Excel workbook, and of a zip container — 

2166#: which is what an .xlsx is (BUG-55). 

2167_OLE2_MAGIC = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1" 

2168_ZIP_MAGIC = b"PK\x03\x04" 

2169 

2170#: Tried in order when a delimited file is not UTF-8 (BUG-55): Windows' default 

2171#: for Western European text first, since that is what Excel writes there, then 

2172#: Latin-1, which decodes any byte and so always ends the search. 

2173_TEXT_ENCODINGS = ("utf-8", "cp1252", "latin-1") 

2174 

2175#: Byte-order marks, longest first (UTF-32 LE's begins with UTF-16 LE's). A 

2176#: file that opens with one is read in that encoding alone: the fallback above 

2177#: would decode EyeLink Data Viewer's UTF-16 export as Latin-1, which never 

2178#: fails, and every column but the first would come out as ``Unnamed: n``. 

2179_BOMS = ( 

2180 (b"\xff\xfe\x00\x00", "utf-32"), 

2181 (b"\x00\x00\xfe\xff", "utf-32"), 

2182 (b"\xef\xbb\xbf", "utf-8-sig"), 

2183 (b"\xff\xfe", "utf-16"), 

2184 (b"\xfe\xff", "utf-16"), 

2185) 

2186 

2187 

2188def _bom_encoding(head: bytes) -> str | None: 

2189 """The encoding a byte-order mark at the start of ``head`` declares.""" 

2190 for bom, encoding in _BOMS: 

2191 if head.startswith(bom): 

2192 return encoding 

2193 return None 

2194 

2195 

2196def _peek(file_like_or_path, size: int = 8) -> bytes: 

2197 """The first ``size`` bytes of an upload or a path, leaving it rewound.""" 

2198 if hasattr(file_like_or_path, "read"): 

2199 _rewind(file_like_or_path) 

2200 head = file_like_or_path.read(size) 

2201 _rewind(file_like_or_path) 

2202 return head if isinstance(head, bytes) else str(head).encode() 

2203 try: 

2204 with open(file_like_or_path, "rb") as handle: 

2205 return handle.read(size) 

2206 except OSError: 

2207 return b"" 

2208 

2209 

2210def _is_workbook(buf, name: str) -> bool: 

2211 """Whether an Excel-named file is a real workbook pandas can open. 

2212 

2213 A zip container is an .xlsx (openpyxl) and an OLE2 container is a legacy 

2214 Excel 97–2003 workbook (xlrd, DATA-53); anything else is delimited text 

2215 wearing an Excel extension (BUG-55). 

2216 """ 

2217 head = _peek(buf) 

2218 return head.startswith((_ZIP_MAGIC, _OLE2_MAGIC)) 

2219 

2220 

2221def _can_reread(buf) -> bool: 

2222 """Whether ``buf`` can be read a second time from the start.""" 

2223 if isinstance(buf, (str, os.PathLike)): 

2224 return True 

2225 seekable = getattr(buf, "seekable", None) 

2226 try: 

2227 return bool(seekable()) if callable(seekable) else False 

2228 except (OSError, ValueError): 

2229 return False 

2230 

2231 

2232def _read_delimited(buf, sep: str, plan: ReadPlan | None, **extra) -> pd.DataFrame: 

2233 """``read_csv`` with the encoding fallback a non-UTF-8 export needs (BUG-55). 

2234 

2235 A CSV saved by Excel on Windows is cp1252, and one umlaut in it made the 

2236 whole upload fail with a raw ``UnicodeDecodeError``. Each encoding in 

2237 ``_TEXT_ENCODINGS`` is tried in turn; a stream that cannot be rewound (a zip 

2238 member) re-raises, and :func:`_read_zipped_table` retries it from memory. 

2239 A byte-order mark settles the encoding outright (a UTF-16 Data Viewer 

2240 export). 

2241 """ 

2242 bom = _bom_encoding(_peek(buf, 4)) if _can_reread(buf) else None 

2243 for encoding in (bom,) if bom else _TEXT_ENCODINGS: 

2244 try: 

2245 return pd.read_csv( 

2246 buf, 

2247 sep=sep, 

2248 low_memory=False, 

2249 encoding=encoding, 

2250 **_read_kwargs(plan), 

2251 **extra, 

2252 ) 

2253 except UnicodeDecodeError: 

2254 if bom or encoding == _TEXT_ENCODINGS[-1] or not _can_reread(buf): 

2255 raise 

2256 _rewind(buf) 

2257 _LOGGER.info( 

2258 "%s is not %s; reading it again as the next encoding", 

2259 getattr(buf, "name", buf), 

2260 encoding, 

2261 ) 

2262 raise AssertionError("unreachable: latin-1 decodes every byte") 

2263 

2264 

2265#: EyeLink writes a value it could not measure as a bare period. Declared 

2266#: missing for *numeric* columns only: a word whose text is "." is a word 

2267#: (PERF-6), so a blanket ``na_values="."`` would delete it from the stimulus. 

2268MISSING_MARKER = "." 

2269 

2270#: Schema fields whose values are numbers, and which therefore read the marker 

2271#: above as missing. The identity fields (participant, trial, text_id, text, 

2272#: screen_id) are deliberately absent — a "." there is a label. 

2273NUMERIC_SCHEMA_FIELDS = frozenset( 

2274 { 

2275 "bottom", 

2276 "canvas_height", 

2277 "canvas_width", 

2278 "duration", 

2279 "fixation_id", 

2280 "height", 

2281 "left", 

2282 "line", 

2283 "right", 

2284 "screen_index", 

2285 "screen_timestamp", 

2286 "timestamp", 

2287 "top", 

2288 "width", 

2289 "word_id", 

2290 "x", 

2291 "y", 

2292 } 

2293) 

2294 

2295#: One number written with a decimal comma (``117,7``) — how a German- or 

2296#: French-locale export writes every fractional value (BUG-54). 

2297_DECIMAL_COMMA = re.compile(r"^[+-]?\d+,\d+$") 

2298#: ...and the shape a *thousands* separator gives the same characters 

2299#: (``1,204``). A column whose every comma looks like this could be either, so it 

2300#: is reported rather than guessed at. 

2301_THOUSANDS_GROUPED = re.compile(r"^[+-]?[1-9]\d{0,2}(,\d{3})+$") 

2302 

2303 

2304def _filled_cells(values: pd.Series) -> pd.Series: 

2305 """The cells of ``values`` that hold something, as stripped text. 

2306 

2307 EyeLink's ``.`` marker and a blank cell are *missing*, not unreadable, so 

2308 they are left out — a planned read already turned them into NaN, and an 

2309 unplanned one must not report them as garbage either. 

2310 """ 

2311 text = values[values.notna()].astype(str).str.strip() 

2312 return text[(text != "") & (text != MISSING_MARKER)] 

2313 

2314 

2315def _unparsed_cells(values: pd.Series, parsed: pd.Series) -> pd.Series: 

2316 """The filled cells of ``values`` that ``parsed`` could not read, as text.""" 

2317 failed = parsed.isna() & values.notna() 

2318 if not failed.any(): 

2319 return pd.Series([], dtype=str) 

2320 return _filled_cells(values[failed]) 

2321 

2322 

2323def _to_number(values: pd.Series) -> pd.Series: 

2324 """``pd.to_numeric(errors="coerce")``, reading a decimal-comma column too. 

2325 

2326 A decimal-comma export (``117,7``) made every fractional cell unparseable, 

2327 and NaN then fell through to the silent fallbacks downstream — each fixation 

2328 snapped to its word's centre, each duration read as 0 (BUG-54). The commas 

2329 are converted only when **every** cell that failed is a decimal-comma number 

2330 and not all of them could be a thousands separator instead; anything else 

2331 stays NaN, for :func:`numeric_parse_issues` to report. 

2332 """ 

2333 parsed = pd.to_numeric(values, errors="coerce") 

2334 if pd.api.types.is_numeric_dtype(values): 

2335 return parsed 

2336 failed = _unparsed_cells(values, parsed) 

2337 if failed.empty or not failed.str.fullmatch(_DECIMAL_COMMA).all(): 

2338 return parsed 

2339 if failed.str.fullmatch(_THOUSANDS_GROUPED).all(): 

2340 return parsed 

2341 converted = pd.to_numeric(failed.str.replace(",", ".", regex=False)) 

2342 parsed = parsed.copy() 

2343 parsed.loc[converted.index] = converted 

2344 return parsed 

2345 

2346 

2347#: How a mapped field's values are read, for :func:`mapping_value_preview`. 

2348_PREVIEW_ID_FIELDS = frozenset( 

2349 { 

2350 "participant", 

2351 "trial", 

2352 "text_id", 

2353 "word_id", 

2354 "fixation_id", 

2355 "screen_id", 

2356 "screen_fixation_id", 

2357 "block", 

2358 } 

2359) 

2360_PREVIEW_TIME_FIELDS = frozenset({"duration", "timestamp", "screen_timestamp"}) 

2361_PREVIEW_PIXEL_FIELDS = frozenset( 

2362 { 

2363 "x", 

2364 "y", 

2365 "width", 

2366 "height", 

2367 "left", 

2368 "right", 

2369 "top", 

2370 "bottom", 

2371 "canvas_width", 

2372 "canvas_height", 

2373 } 

2374) 

2375_PREVIEW_NUMBER_FIELDS = frozenset({"line", "screen_index"}) | { 

2376 key for key, *_ in READING_MEASURE_FIELDS 

2377} 

2378#: Rows looked at — the first few values are all a preview shows, and reading 

2379#: the head keeps it free on a table of millions of rows. 

2380_PREVIEW_ROWS = 200 

2381 

2382 

2383def mapping_value_preview( 

2384 df: pd.DataFrame | None, field_key: str, column, *, limit: int = 3 

2385) -> str: 

2386 """A few of ``column``'s values and what the app reads them as. 

2387 

2388 The mapping editor's value preview: a plausible column name can still hold 

2389 the wrong thing — trial ids picked as a condition, an onset as a duration, 

2390 seconds read as milliseconds — and its first values show it before saving. 

2391 Reads the head of the frame the editor already holds, through the same 

2392 conversions normalization applies (:func:`stable_id`, :func:`_to_number`, 

2393 :func:`time_unit_ms`). ``""`` when there is nothing to show. 

2394 """ 

2395 if df is None or not column: 

2396 return "" 

2397 columns = [str(c) for c in trial_mapping_columns(column)] 

2398 if not columns or any(c not in df.columns for c in columns): 

2399 return "" 

2400 head = df.head(_PREVIEW_ROWS)[columns].dropna(how="all") 

2401 if head.empty: 

2402 return "No values in the first rows" 

2403 

2404 def number(value: float) -> str: 

2405 return f"{value:,.6g}" 

2406 

2407 if len(columns) > 1 or field_key in _PREVIEW_ID_FIELDS: 

2408 ids = trial_id_series(head, column).drop_duplicates().head(limit) 

2409 shown = [] 

2410 for index, value in ids.items(): 

2411 source = " + ".join(str(head.at[index, c]) for c in columns) 

2412 shown.append(value if source == value else f"{source} → {value}") 

2413 return "Read as IDs: " + ", ".join(shown) 

2414 values = head[columns[0]].dropna().head(limit) 

2415 if ( 

2416 field_key in _PREVIEW_TIME_FIELDS 

2417 or field_key in _PREVIEW_PIXEL_FIELDS 

2418 or field_key in _PREVIEW_NUMBER_FIELDS 

2419 ): 

2420 parsed = _to_number(values) 

2421 if parsed.isna().all(): 

2422 return "Not numbers: " + ", ".join(str(v) for v in values) 

2423 factor = 1.0 

2424 unit = "" 

2425 if field_key in _PREVIEW_TIME_FIELDS: 

2426 factor, unit = time_unit_ms(columns[0]), " ms" 

2427 elif field_key in _PREVIEW_PIXEL_FIELDS: 

2428 unit = " px" 

2429 shown = [] 

2430 for source, value in zip(values, parsed): 

2431 if pd.isna(value): 

2432 shown.append(f"{source} (not a number)") 

2433 elif factor != 1.0: 

2434 shown.append(f"{source} → {number(value * factor)}{unit}") 

2435 else: 

2436 shown.append(f"{number(value)}{unit}") 

2437 return ", ".join(shown) 

2438 return ", ".join(f"“{value}”" for value in values.astype(str)) 

2439 

2440 

2441#: What becomes of a fixation or word whose mapped numeric cell is unreadable — 

2442#: said in the warning, because "left empty" means something different per field. 

2443_UNPARSED_CONSEQUENCE = { 

2444 "duration": "those fixations are read as 0 ms long", 

2445 "timestamp": "those fixations are read as starting at 0", 

2446 "x": "those fixations are placed at their word's center when they have a word id, " 

2447 "and left off the plot otherwise", 

2448 "y": "those fixations are placed at their word's center when they have a word id, " 

2449 "and left off the plot otherwise", 

2450 "word_id": "those rows have no word id", 

2451} 

2452#: The same for the Words table, where x/y are a box's corner, not a gaze (#374). 

2453_UNPARSED_WORD_CONSEQUENCE = dict.fromkeys( 

2454 ("x", "y", "width", "height", "left", "right", "top", "bottom"), 

2455 "those word boxes have no position", 

2456) 

2457 

2458 

2459def numeric_parse_issues( 

2460 raw: pd.DataFrame, schema: dict, *, table: str, fixations: bool = True 

2461) -> list[str]: 

2462 """Plain-language warnings for mapped numeric columns that did not parse. 

2463 

2464 One line per column, naming the table, the column, how many of its cells 

2465 were unreadable, a few examples, and what the load did with those rows — 

2466 because the load carries on either way, and a column read as all-NaN used to 

2467 produce a plausible-looking figure with nothing said (BUG-54). A 

2468 decimal-comma column that :func:`_to_number` converts is not an issue; one 

2469 whose commas could equally be thousands separators is, with that named. 

2470 """ 

2471 issues: list[str] = [] 

2472 seen: set = set() 

2473 for key, column in schema.items(): 

2474 if key not in NUMERIC_SCHEMA_FIELDS or not isinstance(column, str): 

2475 continue 

2476 if column in seen or column not in raw.columns: 

2477 continue 

2478 seen.add(column) 

2479 values = raw[column] 

2480 if pd.api.types.is_numeric_dtype(values) or pd.api.types.is_bool_dtype(values): 

2481 continue 

2482 failed = _unparsed_cells(values, _to_number(values)) 

2483 if failed.empty: 

2484 continue 

2485 examples = ", ".join(f"'{v}'" for v in failed.drop_duplicates().head(3)) 

2486 line = ( 

2487 f"{table}: {len(failed):,} of {len(_filled_cells(values)):,} values in " 

2488 f"`{column}` aren't numbers (e.g. {examples})" 

2489 ) 

2490 if failed.str.fullmatch(_THOUSANDS_GROUPED).all(): 

2491 line += ( 

2492 ". They could be a decimal comma or a thousands separator, so they " 

2493 "were not guessed at — re-export the table with a '.' decimal point " 

2494 "and no thousands separator" 

2495 ) 

2496 consequences = ( 

2497 _UNPARSED_CONSEQUENCE if fixations else _UNPARSED_WORD_CONSEQUENCE 

2498 ) 

2499 consequence = consequences.get(key, "those cells are left empty") 

2500 issues.append(f"{line}; {consequence}.") 

2501 return issues 

2502 

2503 

2504def _identity_columns(source: pd.DataFrame, schema: dict) -> list[str]: 

2505 """The source columns a row's (participant, trial) identity is built from.""" 

2506 columns: list = [] 

2507 if schema.get("participant"): 

2508 columns += trial_mapping_columns(schema["participant"]) 

2509 columns += trial_mapping_columns(schema["trial"]) 

2510 return [c for c in dict.fromkeys(columns) if c in source.columns] 

2511 

2512 

2513def _id_missing(values: pd.Series) -> pd.Series: 

2514 """Cells that hold no id: missing, or text that is only whitespace — which 

2515 :func:`stable_id` would turn into an empty id, a trial named ``""``.""" 

2516 missing = values.isna() 

2517 if pd.api.types.is_numeric_dtype(values) or pd.api.types.is_bool_dtype(values): 

2518 return missing 

2519 # Tested per distinct value: an id column has few, and a corpus many rows. 

2520 distinct = pd.Series(values[~missing].unique()) 

2521 blank = distinct[distinct.astype(str).str.strip().eq("")] 

2522 return missing | values.isin(blank) if len(blank) else missing 

2523 

2524 

2525def _rows_missing_identity( 

2526 source: pd.DataFrame, schema: dict 

2527) -> tuple[pd.Series, pd.Series]: 

2528 """``(blank, unkeyed)`` row masks: rows with no participant or trial id. 

2529 

2530 A row missing either cannot belong to any trial, and its NaN crashed the 

2531 load outright — one ``,,,,`` line, the blank row Excel leaves at the end of 

2532 a sheet, made the whole dataset impossible to add (BUG-56). ``blank`` is the 

2533 rows that hold nothing at all, which are not data and go quietly; 

2534 ``unkeyed`` is the rest, which hold data and are reported. An id that is 

2535 only whitespace counts as missing (round 10). Only the missing rows are 

2536 inspected cell by cell, so a clean table costs one ``isna`` and one pass 

2537 over the distinct values per id column. 

2538 """ 

2539 columns = _identity_columns(source, schema) 

2540 missing = source[columns].apply(_id_missing).any(axis=1) if columns else None 

2541 none = pd.Series(False, index=source.index) 

2542 if missing is None or not missing.any(): 

2543 return none, none 

2544 rows = source.loc[missing].drop(columns=[SOURCE_FILE_COLUMN], errors="ignore") 

2545 empty = rows.apply(lambda c: c.isna() | (c.astype(str).str.strip() == "")) 

2546 blank = none.copy() 

2547 blank.loc[rows.index] = empty.all(axis=1) 

2548 return blank, missing & ~blank 

2549 

2550 

2551def _drop_rows_missing_identity(source: pd.DataFrame, schema: dict) -> pd.DataFrame: 

2552 """``source`` without the rows :func:`_rows_missing_identity` flags.""" 

2553 blank, unkeyed = _rows_missing_identity(source, schema) 

2554 keep = ~(blank | unkeyed) 

2555 return source if keep.all() else source.loc[keep] 

2556 

2557 

2558def identity_issues( 

2559 raw: pd.DataFrame, 

2560 schema: dict, 

2561 *, 

2562 table: str, 

2563 unkeyed: pd.Series | None = None, 

2564) -> list[str]: 

2565 """A warning for rows that hold data but no participant or trial id. 

2566 

2567 ``unkeyed`` is :func:`_rows_missing_identity`'s second mask, when the 

2568 caller already has it.""" 

2569 if unkeyed is None: 

2570 _, unkeyed = _rows_missing_identity(raw, schema) 

2571 count = int(unkeyed.sum()) 

2572 if not count: 

2573 return [] 

2574 columns = [ 

2575 c 

2576 for c in _identity_columns(raw, schema) 

2577 if _id_missing(raw.loc[unkeyed, c]).any() 

2578 ] 

2579 named = ", ".join(f"`{c}`" for c in columns) 

2580 if count == 1: 

2581 said = "1 row has no value in {}, so it belongs to no trial and was left out" 

2582 else: 

2583 said = f"{count:,} rows have no value in {{}}, so they belong to no trial and were left out" 

2584 return [f"{table}: {said.format(named)}."] 

2585 

2586 

2587#: Positions that never leave this band are fractions of the screen, not 

2588#: pixels: Gazepoint's FPOGX/FPOGY and Pupil Labs Core's norm_pos_x/y. Wider 

2589#: than 0–1, because a fraction strays a little off-screen. 

2590_FRACTION_BAND = (-0.5, 1.5) 

2591 

2592 

2593def screen_fraction_issues(raw: pd.DataFrame, schema: dict, *, table: str) -> list: 

2594 """A warning when the mapped X/Y are screen fractions rather than pixels. 

2595 

2596 Not converted (DATA-40): the load knows no screen size to scale by, and a 

2597 guessed one would put every fixation in the wrong place while looking 

2598 plausible — the one outcome worse than a figure that is obviously wrong. 

2599 """ 

2600 x, y = schema.get("x"), schema.get("y") 

2601 if not (isinstance(x, str) and isinstance(y, str)): 

2602 return [] 

2603 if x not in raw.columns or y not in raw.columns: 

2604 return [] 

2605 xs, ys = _to_number(raw[x]), _to_number(raw[y]) 

2606 both = xs.notna() & ys.notna() 

2607 if not both.any(): 

2608 return [] 

2609 low, high = _FRACTION_BAND 

2610 xs, ys = xs[both], ys[both] 

2611 if not (xs.between(low, high).all() and ys.between(low, high).all()): 

2612 return [] 

2613 if not (xs.between(0, 1, inclusive="neither").any()): 

2614 return [] # all 0 / all 1 is degenerate data, not a fraction 

2615 return [ 

2616 f"{table}: every position in `{x}` / `{y}` lies between 0 and 1 — these " 

2617 "look like fractions of the screen (Gazepoint's FPOGX/FPOGY, Pupil Labs " 

2618 "Core's norm_pos), not pixels, so the scanpath is drawn in a 1-pixel " 

2619 "corner of the canvas. They are not converted, because the screen size " 

2620 "is not known here: multiply them by the screen width and height in " 

2621 "pixels before uploading (for Pupil Core, whose y points up, use " 

2622 "(1 − y) × height)." 

2623 ] 

2624 

2625 

2626def normalization_issues( 

2627 raw: pd.DataFrame, schema: dict, *, table: str, fixations: bool = False 

2628) -> list[str]: 

2629 """Everything the load will do to ``raw`` under ``schema`` that the user 

2630 should hear about: rows left out for want of an id (BUG-56), mapped numeric 

2631 columns that did not parse (BUG-54), and — for ``fixations``, whose X/Y 

2632 are gaze positions rather than box origins — positions that are screen 

2633 fractions rather than pixels (DATA-40).""" 

2634 issues = identity_issues(raw, schema, table=table) 

2635 issues += numeric_parse_issues(raw, schema, table=table, fixations=fixations) 

2636 if fixations: 

2637 issues += screen_fraction_issues(raw, schema, table=table) 

2638 return issues 

2639 

2640 

2641def _warn_normalization_issues( 

2642 raw: pd.DataFrame, schema: dict, *, table: str, fixations: bool = False 

2643) -> None: 

2644 """Raise each :func:`normalization_issues` line as a ``UserWarning``. 

2645 

2646 The headless API and ``render`` have no page to put a warning on, so the 

2647 normalizers say it themselves; the wizard shows the same lines above 

2648 ✅ Add dataset. 

2649 """ 

2650 for issue in normalization_issues(raw, schema, table=table, fixations=fixations): 

2651 warnings.warn(issue.replace("`", "'"), UserWarning, stacklevel=3) 

2652 

2653 

2654@dataclass(frozen=True) 

2655class ReadPlan: 

2656 """Which columns to parse out of a table, and what counts as missing in them. 

2657 

2658 ``columns`` is ``None`` for "parse everything" — the honest answer when the 

2659 mapping claims nothing in the header, so there is no basis on which to drop 

2660 anything. 

2661 """ 

2662 

2663 columns: tuple[str, ...] | None = None 

2664 na_values: dict[str, list[str]] = field(default_factory=dict) 

2665 #: Columns read as the literal text of each cell, with no cell taken as 

2666 #: missing (BUG-53). The word-text column: "None", "NA" and "null" are 

2667 #: words a stimulus can contain, and pandas' default NA spellings turned 

2668 #: every one of them into NaN before normalization saw it. 

2669 verbatim: tuple[str, ...] = () 

2670 #: Identity columns (participant, trial, text, screen) read as text, so a 

2671 #: zero-padded id survives: CSV inference read `007` as the number 7, while 

2672 #: the same id in a Parquet table stayed "007", and the two tables then 

2673 #: shared no participant at all (BUG-59). Missing cells stay missing. 

2674 identity: tuple[str, ...] = () 

2675 

2676 def narrowed_to(self, available: Iterable[str]) -> ReadPlan: 

2677 """This plan restricted to the columns one file actually has. 

2678 

2679 A multi-file read — several uploads, or several members of one zip — 

2680 plans from the *union* of their headers, and ``usecols`` raises on a 

2681 name the file it is reading does not carry. Narrowing per file is what 

2682 keeps a heterogeneous set readable, exactly as it was before the read 

2683 was planned at all (``read_tables``: "fields absent from a file become 

2684 NaN"). An empty intersection degrades to reading the whole file rather 

2685 than to reading none of it: a file this plan cannot describe is one 

2686 there is no basis to prune. 

2687 """ 

2688 if self.columns is None: 

2689 return self 

2690 present = set(available) 

2691 columns = tuple(name for name in self.columns if name in present) 

2692 verbatim = tuple(c for c in self.verbatim if c in present) 

2693 identity = tuple(c for c in self.identity if c in present) 

2694 if not columns: 

2695 return ReadPlan(verbatim=verbatim, identity=identity) 

2696 return ReadPlan( 

2697 columns=columns, 

2698 na_values={k: v for k, v in self.na_values.items() if k in present}, 

2699 verbatim=verbatim, 

2700 identity=identity, 

2701 ) 

2702 

2703 

2704def _read_kwargs(plan: ReadPlan | None) -> dict: 

2705 """``read_csv`` keywords for a plan (nothing at all for ``None``). 

2706 

2707 An empty ``columns`` reads the whole file, matching the columnar readers in 

2708 :func:`_read_by_extension` — `usecols=[]` would parse nothing at all, and 

2709 the two must not disagree about what an empty plan means. 

2710 """ 

2711 if plan is None: 

2712 return {} 

2713 kwargs: dict = {} 

2714 if plan.columns: 

2715 kwargs["usecols"] = list(plan.columns) 

2716 if plan.na_values: 

2717 kwargs["na_values"] = plan.na_values 

2718 if plan.verbatim: 

2719 # A converter receives the cell's raw text before NA detection runs, and 

2720 # leaves every other column's NA handling exactly as it was (BUG-53). 

2721 kwargs["converters"] = {column: str for column in plan.verbatim} 

2722 if plan.identity: 

2723 kwargs["dtype"] = {column: str for column in plan.identity} 

2724 return kwargs 

2725 

2726 

2727#: pandas' own default missing-value spellings (``read_csv``'s ``na_values`` 

2728#: docs). Spelled out because an Excel read with a verbatim column has to switch 

2729#: the defaults off and hand them back to every *other* column by name. 

2730PANDAS_DEFAULT_NA = frozenset( 

2731 { 

2732 "", 

2733 "#N/A", 

2734 "#N/A N/A", 

2735 "#NA", 

2736 "-1.#IND", 

2737 "-1.#QNAN", 

2738 "-NaN", 

2739 "-nan", 

2740 "1.#IND", 

2741 "1.#QNAN", 

2742 "<NA>", 

2743 "N/A", 

2744 "NA", 

2745 "NULL", 

2746 "NaN", 

2747 "None", 

2748 "n/a", 

2749 "nan", 

2750 "null", 

2751 } 

2752) 

2753 

2754 

2755def _excel_na_kwargs(buf, plan: ReadPlan | None) -> dict: 

2756 """``read_excel`` keywords that keep a plan's verbatim columns literal. 

2757 

2758 ``read_excel`` applies its NA spellings before a converter sees the cell, 

2759 so the CSV path's converter trick does not reach it: the defaults are turned 

2760 off and given back to every other column by name, which needs the header 

2761 first. Excel is never the large-file format, so the second pass is cheap. 

2762 """ 

2763 if plan is None: 

2764 return {} 

2765 kwargs: dict = {} 

2766 if plan.identity: 

2767 kwargs["dtype"] = {column: str for column in plan.identity} 

2768 if not plan.verbatim: 

2769 return kwargs 

2770 header = list(pd.read_excel(buf, nrows=0).columns) 

2771 _rewind(buf) 

2772 na_values = { 

2773 column: sorted(PANDAS_DEFAULT_NA | set(plan.na_values.get(column, ()))) 

2774 for column in header 

2775 if column not in plan.verbatim 

2776 } 

2777 return {**kwargs, "keep_default_na": False, "na_values": na_values} 

2778 

2779 

2780def verbatim_text_plan(header: Sequence[str], schema: dict | None = None) -> ReadPlan: 

2781 """A whole-table words plan: the word text verbatim, the ids as text. 

2782 

2783 For readers that parse every column (the headless API) but still must not 

2784 lose a word spelled "None" or "NA" (BUG-53), nor merge reader ``01`` into 

2785 reader ``1`` by reading the ids as numbers. ``schema`` is the caller's own 

2786 word mapping; without one the columns are auto-detected from the header, 

2787 the way the mapping itself will be. 

2788 """ 

2789 names = list(header) 

2790 schema = schema or propose_word_schema(pd.DataFrame(columns=names)) 

2791 text = schema.get("text") 

2792 verbatim = (text,) if isinstance(text, str) and text in names else () 

2793 return ReadPlan( 

2794 verbatim=verbatim, identity=_identity_columns_in(names, schema, verbatim) 

2795 ) 

2796 

2797 

2798def identity_text_plan( 

2799 header: Sequence[str], schema: dict | None = None, *, kind: str = "fixations" 

2800) -> ReadPlan: 

2801 """A whole-table plan that reads only the identity columns as text. 

2802 

2803 The headless counterpart of what :func:`plan_table_read` does for the app, 

2804 for the tables whose word text is not at stake (fixations, raw gaze): the 

2805 participant / trial / text / screen columns — the caller's ``schema``'s, 

2806 else auto-detected from the header, composite ids expanded — are read as 

2807 text, so ``01`` and ``1`` stay two readers. Every column is still parsed. 

2808 """ 

2809 proposers = { 

2810 "fixations": propose_fix_schema, 

2811 "raw_gaze": propose_raw_gaze_schema, 

2812 } 

2813 if kind not in proposers: 

2814 raise ValueError(f"kind must be one of {sorted(proposers)}, not {kind!r}") 

2815 names = list(header) 

2816 schema = schema or proposers[kind](pd.DataFrame(columns=names)) 

2817 return ReadPlan(identity=_identity_columns_in(names, schema, ())) 

2818 

2819 

2820def _identity_columns_in( 

2821 names: Sequence[str], schema: dict, verbatim: Sequence[str] 

2822) -> tuple[str, ...]: 

2823 """The schema's identity source columns that this header carries.""" 

2824 present = set(names) 

2825 return tuple( 

2826 column 

2827 for column in dict.fromkeys( 

2828 [*_schema_identity_columns(schema), *_IDENTITY_SOURCES] 

2829 ) 

2830 if column in present and column not in verbatim and column not in _ORDINALS 

2831 ) 

2832 

2833 

2834def plan_table_read( 

2835 header: Sequence[str], 

2836 schema: dict, 

2837 registry: Sequence[tuple], 

2838 *, 

2839 filter_fields: Iterable[str] | None = None, 

2840 keep_columns: Iterable[str] | None = None, 

2841 text_column: str | None = None, 

2842 identity_columns: Iterable[str] = (), 

2843) -> ReadPlan: 

2844 """Narrow a read to the columns ``normalize_*`` keeps (PERF-6). 

2845 

2846 An EyeLink IA report ships 174 columns and a fixation report 300; the 

2847 mapping plus ``*_OPTIONAL_FIELDS`` claim 27 and 17, and normalization drops 

2848 the rest on the next line. Parsing them anyway is the largest single cost on 

2849 the load path — measured on the full OneStop reports, planning the read 

2850 takes it from ~200 s / 25 GB to ~58 s / 2.2 GB for a byte-identical frame. 

2851 

2852 ``header`` is the column names alone (see :func:`read_table_columns`), so 

2853 the plan is made before a single row is parsed. ``filter_fields`` and 

2854 ``keep_columns`` carry the columns the user chose to keep beyond the 

2855 mapping, exactly as :func:`compute_keep_columns` takes them. 

2856 

2857 The word-text column — ``text_column`` when the user has mapped one by 

2858 hand, else the schema's own ``text`` — is read verbatim (BUG-53), and the 

2859 identity columns — the schema's, plus any ``identity_columns`` the user 

2860 picked by hand — as text (BUG-59). 

2861 """ 

2862 names = list(header) 

2863 present = set(names) 

2864 text = text_column or schema.get("text") 

2865 verbatim = (text,) if isinstance(text, str) and text in present else () 

2866 identity = tuple( 

2867 column 

2868 for column in dict.fromkeys( 

2869 [*_schema_identity_columns(schema), *identity_columns, *_IDENTITY_SOURCES] 

2870 ) 

2871 if column in present and column not in verbatim and column not in _ORDINALS 

2872 ) 

2873 if not (set(_schema_source_columns(schema)) & present): 

2874 # Nothing is mapped yet — an unmapped upload, or a table this schema 

2875 # does not describe. Dropping columns here would be guessing. 

2876 return ReadPlan(verbatim=verbatim, identity=identity) 

2877 keep = compute_keep_columns( 

2878 schema, 

2879 optional_sources=[row[0] for row in registry if row[0] in present], 

2880 filter_fields=filter_fields, 

2881 keep_columns=set(keep_columns or ()) | set(verbatim), 

2882 ) 

2883 numeric = {row[0] for row in registry if row[2] == "numeric"} 

2884 numeric |= { 

2885 column 

2886 for key, column in schema.items() 

2887 if key in NUMERIC_SCHEMA_FIELDS and isinstance(column, str) 

2888 } 

2889 columns = tuple(name for name in names if name in keep) 

2890 return ReadPlan( 

2891 columns=columns, 

2892 na_values={name: [MISSING_MARKER] for name in columns if name in numeric}, 

2893 verbatim=verbatim, 

2894 identity=tuple(c for c in identity if c in keep), 

2895 ) 

2896 

2897 

2898#: Schema fields that name *which* participant / trial / text / screen a row 

2899#: belongs to — read as text, never as numbers (BUG-59). 

2900IDENTITY_SCHEMA_FIELDS = ("participant", "trial", "text_id", "screen_id") 

2901#: The id columns `normalize_*` consults by name rather than through the schema. 

2902_IDENTITY_SOURCES = ("unique_trial_id", "unique_paragraph_id") 

2903#: ...except an index that is also carried as a number: the trial picker sorts 

2904#: on `TRIAL_INDEX`, and as text 10 would sort before 2. 

2905_ORDINALS = frozenset({"TRIAL_INDEX", "trial_index"}) 

2906 

2907 

2908def _schema_identity_columns(schema: dict) -> list[str]: 

2909 """The source columns a schema's identity fields name (lists expanded).""" 

2910 columns: list = [] 

2911 for key in IDENTITY_SCHEMA_FIELDS: 

2912 value = schema.get(key) 

2913 if value: 

2914 columns += trial_mapping_columns(value) 

2915 return columns 

2916 

2917 

2918def read_table_columns(file_like_or_path, *, kind: str | None = None) -> list[str]: 

2919 """A table's column names, without parsing its rows. 

2920 

2921 The header pass that :func:`plan_table_read` plans from. Delimited text 

2922 reads zero rows and columnar formats read their schema; anything else 

2923 (Excel, a zip of several members) falls back to reading the table, because 

2924 there is no cheaper way to learn its columns and those formats are not the 

2925 ones that hurt. ``kind`` (``"words"`` / ``"fixations"``) picks a mixed 

2926 zip's members the way :func:`zip_member_split` does (#374, F3). 

2927 """ 

2928 name = getattr(file_like_or_path, "name", str(file_like_or_path)).lower() 

2929 _rewind(file_like_or_path) 

2930 try: 

2931 if name.endswith(".zip"): 

2932 return _zipped_table_columns(file_like_or_path, kind=kind) 

2933 if name.endswith(".parquet"): 

2934 import pyarrow.parquet as pq 

2935 

2936 return list(pq.read_schema(file_like_or_path).names) 

2937 if name.endswith(".feather"): 

2938 from pyarrow import feather 

2939 

2940 return list(feather.read_table(file_like_or_path, columns=[]).schema.names) 

2941 if name.endswith((".tsv", ".tab", ".csv", ".txt")): 

2942 return _header(file_like_or_path, _sniff_delimiter(file_like_or_path, name)) 

2943 # Rewound afterwards too (the `finally`): the caller reads the table 

2944 # again, and a buffer left at its end reads as an empty file. 

2945 return list(read_table(file_like_or_path).columns) 

2946 except pd.errors.EmptyDataError as exc: 

2947 raise _empty_file_error(name) from exc 

2948 finally: 

2949 _rewind(file_like_or_path) 

2950 

2951 

2952def _header(buf, sep: str) -> list[str]: 

2953 """A delimited table's column names, read under the encoding fallback.""" 

2954 return list(_read_delimited(buf, sep, None, nrows=0).columns) 

2955 

2956 

2957def _empty_file_error(name: str) -> ValueError: 

2958 """The error an empty upload raises, in place of pandas' "No columns to 

2959 parse from file" (BUG-55).""" 

2960 return ValueError( 

2961 f"'{Path(name).name}' is empty — it has no header row. Check the export " 

2962 "finished writing, then upload it again." 

2963 ) 

2964 

2965 

2966#: How far into a zip member the header line is looked for. 

2967_HEADER_MAX_BYTES = 1024 * 1024 

2968 

2969 

2970def _member_layout(zf: zipfile.ZipFile, info: zipfile.ZipInfo) -> tuple[list, str]: 

2971 """Column names and delimiter of one delimited zip member, from its first 

2972 line alone. 

2973 

2974 A member stream cannot be rewound, so the encoding fallback runs over just 

2975 the header line held in memory — cut at the newline, never mid-character — 

2976 and the delimiter (DATA-41) is read off the same line. 

2977 """ 

2978 with zf.open(info) as inner: 

2979 head = inner.read(_HEADER_MAX_BYTES) 

2980 line = _first_line(head) 

2981 if not line.strip(): 

2982 raise _empty_file_error(info.filename) 

2983 sep = _delimiter_of(line, info.filename) 

2984 return _header(io.BytesIO(line + b"\n"), sep), sep 

2985 

2986 

2987def _zipped_table_columns(file_like_or_path, *, kind: str | None = None) -> list[str]: 

2988 """Column names of a ``.zip``'s members, without decompressing the rows. 

2989 

2990 A OneStop report is a single ~4 GB CSV inside its zip, so the fallback of 

2991 reading the table and taking its columns would unpack the whole archive to 

2992 answer a question the first line already does. Members are unioned in order, 

2993 matching how :func:`_read_zipped_table` concatenates them. 

2994 """ 

2995 columns: list[str] = [] 

2996 with zipfile.ZipFile(file_like_or_path) as zf: 

2997 infos = [ 

2998 i 

2999 for i in zf.infolist() 

3000 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__")) 

3001 ] 

3002 if not infos: 

3003 raise ValueError( 

3004 "the zip holds no table file (CSV, TSV, TXT, Excel, Parquet or Feather)" 

3005 ) 

3006 # DATA-16/S6: the same declared-size guard the full read applies. A 

3007 # cheaper way to learn an archive's columns must not also be a way 

3008 # around its decompression limits. 

3009 _check_zip_limits(infos) 

3010 infos = _zip_split(zf, infos, kind).used_infos 

3011 for info in infos: 

3012 name = info.filename.lower() 

3013 if not name.endswith((".tsv", ".tab", ".csv", ".txt")): 

3014 # Columnar/workbook members seek, so there is no header-only 

3015 # read: defer to `_read_zipped_table`, which reads every member 

3016 # under a running byte budget (declared sizes are forgeable, so 

3017 # the check above is not sufficient on its own). 

3018 return list(_read_zipped_table(file_like_or_path, kind=kind).columns) 

3019 names, _sep = _member_layout(zf, info) 

3020 columns.extend(c for c in names if c not in columns) 

3021 return columns 

3022 

3023 

3024def _rewind(file_like_or_path) -> None: 

3025 """Seek an uploaded buffer back to the start, so it can be read again.""" 

3026 seek = getattr(file_like_or_path, "seek", None) 

3027 if callable(seek): 

3028 seek(0) 

3029 

3030 

3031def read_mapped_table( 

3032 file_like_or_path, 

3033 *, 

3034 kind: str, 

3035 declared: dict | None = None, 

3036 filter_fields: Iterable[str] | None = None, 

3037) -> pd.DataFrame: 

3038 """Read one words or fixations table, parsing only the columns it needs. 

3039 

3040 The header-then-plan-then-read pass in one call — what a loader that already 

3041 knows which table it is reading wants (PERF-6). ``declared`` is a publisher's 

3042 own mapping, used in place of auto-detection; ``filter_fields`` are extra 

3043 source columns to keep for trial filtering. 

3044 

3045 The frame comes back *pre-normalization*, exactly as a plain 

3046 :func:`read_table` would return it, minus the columns nothing claims. 

3047 """ 

3048 proposers = { 

3049 "words": (propose_word_schema, WORD_OPTIONAL_FIELDS), 

3050 "fixations": (propose_fix_schema, FIX_OPTIONAL_FIELDS), 

3051 } 

3052 if kind not in proposers: 

3053 raise ValueError(f"kind must be one of {sorted(proposers)}, not {kind!r}") 

3054 propose, registry = proposers[kind] 

3055 header = read_table_columns(file_like_or_path) 

3056 schema = dict(declared) if declared else propose(pd.DataFrame(columns=header)) 

3057 plan = plan_table_read(header, schema, registry, filter_fields=filter_fields) 

3058 return read_table(file_like_or_path, plan=plan) 

3059 

3060 

3061def source_labels(paths: Sequence[str]) -> list[str]: 

3062 """The ``source_file`` label for each path — its stem, unless that stem 

3063 is shared with another path. 

3064 

3065 ``reader-a/fixations.csv`` and ``reader-b/fixations.csv`` both have the 

3066 stem ``fixations``, and a label is mapped as participant or trial identity, 

3067 so two readers would silently become one. A shared stem is qualified by 

3068 the fewest trailing folders that tell it apart from the paths it clashes 

3069 with (``reader-a/fixations``) — only folders that *differ*, so the shared 

3070 part of an absolute path (``/Users/<name>/…``) never enters a label. Paths 

3071 in the same folder then keep their extension (``fix.csv`` / ``fix.tsv``), 

3072 and paths that are the same get a ``#n`` occurrence number. Backslashes 

3073 count as folder separators, so a zip made on Windows labels as one made 

3074 elsewhere. 

3075 

3076 A browser upload carries no folders, so two same-named uploads read 

3077 ``fixations#1`` / ``fixations#2`` in the app where the same files read from 

3078 disk (API, CLI) get their folders. :func:`source_file_name` recovers the 

3079 file name from any of these forms. 

3080 """ 

3081 split = [] 

3082 for path in paths: 

3083 parts = [ 

3084 p for p in str(path).replace("\\", "/").split("/") if p not in ("", ".") 

3085 ] 

3086 parts = parts or [str(path)] 

3087 split.append((tuple(parts[:-1]), parts[-1])) 

3088 stems = [Path(name).stem for _, name in split] 

3089 

3090 def shared_tail(a: tuple, b: tuple) -> int: 

3091 n = 0 

3092 while n < min(len(a), len(b)) and a[-1 - n] == b[-1 - n]: 

3093 n += 1 

3094 return n 

3095 

3096 labels = list(stems) 

3097 for i, (folders, name) in enumerate(split): 

3098 clashes = [j for j, stem in enumerate(stems) if j != i and stem == stems[i]] 

3099 if not clashes: 

3100 continue 

3101 others = [split[j][0] for j in clashes if split[j][0] != folders] 

3102 depth = min( 

3103 len(folders), max((shared_tail(folders, o) + 1 for o in others), default=0) 

3104 ) 

3105 same_folder = any(split[j][0] == folders for j in clashes) 

3106 tail = Path(name).name if same_folder else stems[i] 

3107 if same_folder and any(split[j] == split[i] for j in clashes): 

3108 tail = stems[i] 

3109 labels[i] = "/".join([*folders[len(folders) - depth :], tail]) 

3110 seen: dict[str, int] = {} 

3111 counts: dict[str, int] = {} 

3112 for label in labels: 

3113 counts[label] = counts.get(label, 0) + 1 

3114 for i, label in enumerate(labels): 

3115 if counts[label] > 1: 

3116 seen[label] = seen.get(label, 0) + 1 

3117 labels[i] = f"{label}#{seen[label]}" 

3118 return labels 

3119 

3120 

3121def source_file_name(label: str) -> str: 

3122 """The file stem inside a :func:`source_labels` label — without the folders 

3123 that qualify it, its ``#n`` occurrence number or a kept extension — for 

3124 code that parses identity out of a file's name (MultiplEYE uploads).""" 

3125 name = re.sub(r"#\d+$", "", str(label).rsplit("/", 1)[-1]) 

3126 stem = Path(name).stem 

3127 return stem if Path(name).suffix.lower().lstrip(".") in UPLOAD_FILE_TYPES else name 

3128 

3129 

3130def _tag_and_concat( 

3131 frames: list[pd.DataFrame], 

3132 labels: list[str], 

3133 source_column: str | None, 

3134 *, 

3135 always_tag: bool = False, 

3136) -> pd.DataFrame: 

3137 """Concatenate frames into one, tagging each with its source label in 

3138 ``source_column`` (unless that frame already carries the column, or 

3139 ``source_column`` is None) so rows stay traceable to their origin. 

3140 

3141 By default only multi-frame reads are tagged (a lone frame needs no origin 

3142 marker). ``always_tag=True`` tags a single frame too — used by 

3143 :func:`read_tables` so a one-file upload still exposes its filename (as 

3144 ``source_file``), which the upload wizard can map as the trial / participant 

3145 id when no column carries it. Columns are aligned by name; fields absent 

3146 from a frame become NaN for its rows.""" 

3147 if source_column and (always_tag or len(frames) > 1): 

3148 for df, label in zip(frames, labels): 

3149 if source_column not in df.columns: 

3150 df[source_column] = label 

3151 if len(frames) == 1: 

3152 return frames[0] 

3153 return pd.concat(frames, ignore_index=True, sort=False) 

3154 

3155 

3156# DATA-16 (security review S6): bound zip decompression. `_read_zipped_table` 

3157# used to `read()` every member with no ceiling, so a small archive of highly 

3158# compressible CSV could expand to many gigabytes and OOM-kill the process — on 

3159# the ~1 GB hosted demo that takes every concurrent visitor's session with it. 

3160# The limits are deliberately generous: real eye-tracking exports are large (the 

3161# OneStop reports are hundreds of MB to a few GB of CSV per zipped table), so the 

3162# absolute caps only catch the honestly-enormous case, and the *ratio* is what 

3163# actually distinguishes a zip bomb from a big corpus. 

3164# 

3165# DATA-34 raised the absolute caps (4/8 GB → 32/64 GB) and made all three 

3166# tunable: a full OneStop fixation report is a single ~8 GB CSV inside its zip, 

3167# which the old per-member cap refused outright on a workstation with the memory 

3168# to read it. The caps are a memory guard, not a security boundary — the ratio 

3169# check is what discriminates a bomb — so the default now suits the machine that 

3170# actually holds a corpus, and a memory-capped deployment tightens it with the 

3171# env vars below (read once, at import). 

3172ZIP_MAX_MEMBER_ENV = "SCANPATH_ZIP_MAX_MEMBER_GB" 

3173ZIP_MAX_TOTAL_ENV = "SCANPATH_ZIP_MAX_TOTAL_GB" 

3174ZIP_MAX_RATIO_ENV = "SCANPATH_ZIP_MAX_RATIO" 

3175 

3176 

3177def _zip_limit_from_env(var: str, default: float) -> float: 

3178 """Read a zip limit from the environment, falling back to ``default``. 

3179 

3180 The value is returned in whatever unit the caller works in (the two size 

3181 vars are read as gigabytes and scaled here; the ratio is unitless). An 

3182 unset, unparseable or non-positive value keeps the default rather than 

3183 failing the import — a typo in a deployment's env should not stop the app 

3184 from reading data.""" 

3185 raw = os.environ.get(var, "").strip() 

3186 if not raw: 

3187 return default 

3188 try: 

3189 value = float(raw) 

3190 except ValueError: 

3191 return default 

3192 return value if value > 0 else default 

3193 

3194 

3195ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES = int( 

3196 _zip_limit_from_env(ZIP_MAX_MEMBER_ENV, 32.0) * 1024**3 

3197) # any single member 

3198ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES = int( 

3199 _zip_limit_from_env(ZIP_MAX_TOTAL_ENV, 64.0) * 1024**3 

3200) # across the whole archive 

3201ZIP_MAX_COMPRESSION_RATIO = _zip_limit_from_env( 

3202 ZIP_MAX_RATIO_ENV, 200.0 

3203) # uncompressed / compressed 

3204# Below this the ratio is not checked at all: a small but very repetitive table 

3205# can legitimately compress 1000×, and expanding it costs nothing. 

3206ZIP_RATIO_CHECK_MIN_BYTES = 256 * 1024 * 1024 

3207 

3208 

3209def _format_bytes(n: float) -> str: 

3210 """Human-readable byte size for a user-facing error message.""" 

3211 for unit in ("B", "KB", "MB", "GB"): 

3212 if abs(n) < 1024 or unit == "GB": 

3213 return f"{n:.0f} {unit}" if unit == "B" else f"{n:.1f} {unit}" 

3214 n /= 1024.0 

3215 return f"{n:.1f} GB" 

3216 

3217 

3218def _check_zip_limits(infos: list[zipfile.ZipInfo]) -> None: 

3219 """Reject an archive that would decompress past the DATA-16 limits. 

3220 

3221 Checks the *declared* sizes (``ZipInfo.file_size``) before a single member is 

3222 opened, so an oversized archive fails fast instead of being discovered by 

3223 exhausting RAM. Declared sizes can be forged, so :func:`_read_zipped_table` 

3224 additionally reads each member under a hard byte budget. 

3225 """ 

3226 total = sum(int(i.file_size) for i in infos) 

3227 compressed = sum(int(i.compress_size) for i in infos) 

3228 for info in infos: 

3229 if int(info.file_size) > ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES: 

3230 raise ValueError( 

3231 f"{info.filename!r} in the zip unpacks to " 

3232 f"{_format_bytes(info.file_size)}, over the " 

3233 f"{_format_bytes(ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES)} per-file " 

3234 "limit. Split it into smaller files (one per participant works " 

3235 f"well). Running it yourself? {ZIP_MAX_MEMBER_ENV} (GB) raises it." 

3236 ) 

3237 if total > ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES: 

3238 raise ValueError( 

3239 f"the zip unpacks to {_format_bytes(total)}, over the " 

3240 f"{_format_bytes(ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES)} limit. Split it " 

3241 "into smaller uploads. Running it yourself? " 

3242 f"{ZIP_MAX_TOTAL_ENV} (GB) raises it." 

3243 ) 

3244 if total >= ZIP_RATIO_CHECK_MIN_BYTES: 

3245 ratio = total / max(compressed, 1) 

3246 if ratio > ZIP_MAX_COMPRESSION_RATIO: 

3247 raise ValueError( 

3248 f"the zip archive expands {ratio:.0f}× (to " 

3249 f"{_format_bytes(total)}), above the " 

3250 f"{ZIP_MAX_COMPRESSION_RATIO:.0f}× limit — it doesn't look like " 

3251 "a normal data export. Unzip it and load the tables directly if " 

3252 "this is genuine." 

3253 ) 

3254 

3255 

3256#: Formats whose pandas reader seeks (columnar footers, zip-container 

3257#: workbooks), so their member can't be streamed and is read into memory whole. 

3258_SEEKABLE_ONLY_SUFFIXES = (".parquet", ".feather", ".xlsx", ".xls") 

3259 

3260 

3261class _BudgetedZipMember(io.RawIOBase): 

3262 """One zip member, readable only up to ``budget`` bytes. 

3263 

3264 A zip's central directory is attacker-controlled, so the declared sizes 

3265 :func:`_check_zip_limits` reads can be forged; this counts the bytes that 

3266 actually arrive and raises as soon as they pass the budget. Streaming them 

3267 (rather than ``read()``-ing the member into one big ``bytes`` first) also 

3268 keeps an honestly enormous CSV — a full OneStop fixation report is ~8 GB 

3269 inside its zip — from needing a second full-size copy of itself in memory 

3270 before pandas sees a byte of it (DATA-34). 

3271 """ 

3272 

3273 def __init__( 

3274 self, 

3275 inner, 

3276 budget: int, 

3277 name: str, 

3278 *, 

3279 limit_label: str = "archive decompression limit", 

3280 ) -> None: 

3281 self._inner = inner 

3282 self._budget = int(budget) 

3283 self._name = name 

3284 self._limit_label = limit_label 

3285 self.consumed = 0 

3286 

3287 def readable(self) -> bool: 

3288 return True 

3289 

3290 def readinto(self, buffer) -> int: 

3291 read = self._inner.readinto(buffer) 

3292 if not read: 

3293 return 0 

3294 self.consumed += read 

3295 if self.consumed > self._budget: 

3296 raise ValueError( 

3297 f"{self._name!r} in the zip is larger than the archive says, " 

3298 f"past the {_format_bytes(self._budget)} {self._limit_label}, so " 

3299 "it was not read. Re-create the zip and upload it again." 

3300 ) 

3301 return read 

3302 

3303 

3304#: What each table kind's members are called in the mixed-zip message (#374 F3). 

3305_ZIP_KIND_NOUNS = { 

3306 "fixations": ("fixation report", "fixation reports"), 

3307 "words": ("interest-area report", "interest-area reports"), 

3308} 

3309#: The wizard row each kind belongs in, as the message names it. 

3310_ZIP_KIND_ROWS = {"fixations": "Fixations", "words": "Words"} 

3311 

3312 

3313def table_kind_of_columns(columns: Iterable[str]) -> str | None: 

3314 """Whether a header looks like a fixation table or a word (interest-area) 

3315 table — ``"fixations"``, ``"words"`` or ``None`` when it is neither or both. 

3316 

3317 Reuses the mapping's own detection: a fixation table has a duration column 

3318 and no word box, a word table has a word box and no fixation duration. An 

3319 EyeLink Fixation Report and Interest Area Report land on opposite sides. 

3320 """ 

3321 frame = pd.DataFrame(columns=list(dict.fromkeys(map(str, columns)))) 

3322 words = propose_word_schema(frame) 

3323 has_box = all(words.get(k) for k in ("x", "y", "width", "height")) or all( 

3324 words.get(k) for k in _BOX_EDGES 

3325 ) 

3326 has_duration = bool(propose_fix_schema(frame).get("duration")) 

3327 if has_duration and not has_box: 

3328 return "fixations" 

3329 if has_box and not has_duration: 

3330 return "words" 

3331 return None 

3332 

3333 

3334@dataclass(frozen=True) 

3335class ZipMemberSplit: 

3336 """Which members of a zip one table reads (#374 F3). 

3337 

3338 A ZIP of per-participant folders commonly holds both EyeLink reports. Its 

3339 members are grouped by column set, each set classified by 

3340 :func:`table_kind_of_columns`; when the archive holds both kinds, the table 

3341 being read keeps the members of its own kind and the rest are *left out* — 

3342 never concatenated into one frame, which put a "fixation" at every word. 

3343 """ 

3344 

3345 used: tuple[str, ...] 

3346 left_out: tuple[str, ...] = () 

3347 kind: str | None = None 

3348 left_out_kind: str | None = None 

3349 used_infos: tuple = () 

3350 

3351 @property 

3352 def mixed(self) -> bool: 

3353 return bool(self.left_out) 

3354 

3355 def message(self) -> str: 

3356 """The one-line note the upload row shows ("" when nothing was left out).""" 

3357 if not self.left_out or self.kind not in _ZIP_KIND_NOUNS: 

3358 return "" 

3359 n_used, n_out = len(self.used), len(self.left_out) 

3360 noun = _ZIP_KIND_NOUNS[self.kind][n_used != 1] 

3361 if self.left_out_kind in _ZIP_KIND_NOUNS: 

3362 other = _ZIP_KIND_NOUNS[self.left_out_kind][n_out != 1] 

3363 row = _ZIP_KIND_ROWS[self.left_out_kind] 

3364 tail = f" — add the ZIP to the {row} row too." 

3365 else: 

3366 other, tail = ("other file" if n_out == 1 else "other files"), "." 

3367 verb = "was" if n_out == 1 else "were" 

3368 return ( 

3369 f"Using the {n_used} {noun} in this ZIP; {n_out} {other} {verb} " 

3370 f"left out{tail}" 

3371 ) 

3372 

3373 

3374def _zip_member_columns(zf: zipfile.ZipFile, info: zipfile.ZipInfo) -> list[str]: 

3375 """One member's column names — the header line for delimited text, else 

3376 the member read (under the per-file budget) and its schema taken.""" 

3377 name = info.filename.lower() 

3378 if name.endswith((".tsv", ".tab", ".csv", ".txt")): 

3379 return _member_layout(zf, info)[0] 

3380 with zf.open(info) as inner: 

3381 stream = _BudgetedZipMember( 

3382 inner, ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES, info.filename 

3383 ) 

3384 buf = io.BytesIO(stream.read()) 

3385 buf.name = info.filename 

3386 return read_table_columns(buf) 

3387 

3388 

3389def _zip_split( 

3390 zf: zipfile.ZipFile, infos: list[zipfile.ZipInfo], kind: str | None 

3391) -> ZipMemberSplit: 

3392 """Split ``infos`` into the members ``kind`` reads and those it leaves out.""" 

3393 every = ZipMemberSplit( 

3394 used=tuple(i.filename for i in infos), kind=kind, used_infos=tuple(infos) 

3395 ) 

3396 if kind not in _ZIP_KIND_NOUNS or len(infos) < 2: 

3397 return every 

3398 by_columns: dict[tuple, str | None] = {} 

3399 kinds = [] 

3400 for info in infos: 

3401 columns = tuple(_zip_member_columns(zf, info)) 

3402 if columns not in by_columns: 

3403 by_columns[columns] = table_kind_of_columns(columns) 

3404 kinds.append(by_columns[columns]) 

3405 present = {k for k in kinds if k} 

3406 if len(present) < 2 or kind not in present: 

3407 return every 

3408 used = [i for i, k in zip(infos, kinds) if k == kind] 

3409 out_kinds = {k for k in kinds if k != kind} 

3410 return ZipMemberSplit( 

3411 used=tuple(i.filename for i in used), 

3412 left_out=tuple(i.filename for i, k in zip(infos, kinds) if k != kind), 

3413 kind=kind, 

3414 left_out_kind=out_kinds.pop() if len(out_kinds) == 1 else None, 

3415 used_infos=tuple(used), 

3416 ) 

3417 

3418 

3419def zip_member_split(file_like_or_path, kind: str | None) -> ZipMemberSplit: 

3420 """Which members of a ``.zip`` the ``kind`` table reads, and which it 

3421 leaves out (#374 F3) — what the upload row reports. A non-zip, or a zip 

3422 holding one kind of table, uses everything.""" 

3423 name = getattr(file_like_or_path, "name", str(file_like_or_path)) 

3424 if not name.lower().endswith(".zip"): 

3425 return ZipMemberSplit(used=(name,), kind=kind) 

3426 _rewind(file_like_or_path) 

3427 try: 

3428 with zipfile.ZipFile(file_like_or_path) as zf: 

3429 infos = [ 

3430 i 

3431 for i in zf.infolist() 

3432 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__")) 

3433 ] 

3434 _check_zip_limits(infos) 

3435 return _zip_split(zf, infos, kind) 

3436 finally: 

3437 _rewind(file_like_or_path) 

3438 

3439 

3440#: Rows :func:`read_table_sample` parses, and the bytes it reads of a zip member. 

3441_SAMPLE_ROWS = 2000 

3442_SAMPLE_BYTES = 4 * 1024 * 1024 

3443 

3444 

3445def read_table_sample( 

3446 file_like_or_path, *, kind: str | None = None, nrows: int = _SAMPLE_ROWS 

3447) -> pd.DataFrame: 

3448 """The first rows of a delimited table, every column parsed (#374 F13). 

3449 

3450 What the wizard judges an unmapped column's values by, when its planned 

3451 read (PERF-6) left that column out. Delimited text only — a plain file or 

3452 a zip's first member of ``kind`` (:func:`zip_member_split`); anything else, 

3453 or a file that will not parse, gives an empty frame, which the caller reads 

3454 as "no evidence". 

3455 """ 

3456 name = getattr(file_like_or_path, "name", str(file_like_or_path)).lower() 

3457 delimited = (".tsv", ".tab", ".csv", ".txt") 

3458 _rewind(file_like_or_path) 

3459 try: 

3460 if name.endswith(delimited): 

3461 sep = _sniff_delimiter(file_like_or_path, name) 

3462 return _read_delimited(file_like_or_path, sep, None, nrows=nrows) 

3463 if not name.endswith(".zip"): 

3464 return pd.DataFrame() 

3465 with zipfile.ZipFile(file_like_or_path) as zf: 

3466 infos = [ 

3467 i 

3468 for i in zf.infolist() 

3469 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__")) 

3470 ] 

3471 _check_zip_limits(infos) 

3472 used = _zip_split(zf, infos, kind).used_infos 

3473 info = next( 

3474 (i for i in used if i.filename.lower().endswith(delimited)), None 

3475 ) 

3476 if info is None: 

3477 return pd.DataFrame() 

3478 sep = _member_layout(zf, info)[1] 

3479 with zf.open(info) as inner: 

3480 head = inner.read(_SAMPLE_BYTES) 

3481 # Cut at the last whole line: the read stops mid-row. 

3482 if len(head) == _SAMPLE_BYTES and b"\n" in head: 

3483 head = head[: head.rindex(b"\n") + 1] 

3484 return _read_delimited(io.BytesIO(head), sep, None, nrows=nrows) 

3485 except Exception: # no evidence, not an error: the real read reports it 

3486 return pd.DataFrame() 

3487 finally: 

3488 _rewind(file_like_or_path) 

3489 

3490 

3491def looks_like_condition(values: pd.Series) -> bool: 

3492 """Whether a column reads as a condition or an item id — a few repeated 

3493 values, such as ``Adv`` / ``Ele`` or ``2_1`` … ``2_12`` (#374 F13). 

3494 

3495 Between 2 and 50 distinct values, each used at least twice on average; a 

3496 fractional number is a measurement, never a condition.""" 

3497 filled = values.dropna() 

3498 if filled.empty: 

3499 return False 

3500 if pd.api.types.is_float_dtype(filled) and not (filled % 1 == 0).all(): 

3501 return False 

3502 distinct = filled.nunique() 

3503 return 2 <= distinct <= 50 and distinct * 2 <= len(filled) 

3504 

3505 

3506def _read_zipped_table( 

3507 file_like_or_path, *, plan: ReadPlan | None = None, kind: str | None = None 

3508) -> pd.DataFrame: 

3509 """Read table(s) from a ``.zip`` archive (e.g. ``data.csv.zip``). 

3510 

3511 Each member is dispatched on its own extension, so a zip may wrap any 

3512 supported format. A multi-member archive is concatenated just like a 

3513 multi-file upload — every member's rows tagged with its stem in 

3514 ``source_file`` (qualified by its folders when two members share a stem, 

3515 :func:`source_labels`). With ``kind``, an archive that mixes fixation and 

3516 interest-area reports keeps only the members that fit it 

3517 (:func:`zip_member_split`, #374 F3). pandas infers compression only from string paths, not from 

3518 uploaded file-like objects, so we open the archive ourselves. Raises 

3519 ``ValueError`` if the archive holds no data file (macOS ``__MACOSX``/dotfile 

3520 cruft is ignored), or if it would decompress past the DATA-16 size limits 

3521 (``ZIP_MAX_*``) — both the declared sizes and the bytes actually read are 

3522 bounded, so a forged header can't slip past.""" 

3523 with zipfile.ZipFile(file_like_or_path) as zf: 

3524 infos = [ 

3525 i 

3526 for i in zf.infolist() 

3527 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__")) 

3528 ] 

3529 if not infos: 

3530 raise ValueError( 

3531 "the zip holds no table file (CSV, TSV, TXT, Excel, Parquet or Feather)" 

3532 ) 

3533 _check_zip_limits(infos) 

3534 infos = _zip_split(zf, infos, kind).used_infos 

3535 remaining = ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES 

3536 frames, labels = [], [] 

3537 for info in infos: 

3538 member = info.filename 

3539 name = member.lower() 

3540 member_budget = min(remaining, ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES) 

3541 limit_label = ( 

3542 "per-file limit" 

3543 if ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES <= remaining 

3544 else "archive decompression limit" 

3545 ) 

3546 with zf.open(info) as inner: 

3547 stream = _BudgetedZipMember( 

3548 inner, 

3549 member_budget, 

3550 member, 

3551 limit_label=limit_label, 

3552 ) 

3553 if name.endswith(_SEEKABLE_ONLY_SUFFIXES): 

3554 # Columnar/workbook readers seek, so these still land in 

3555 # memory whole — bounded by the same budget. 

3556 buf = io.BytesIO(stream.read()) 

3557 # BUG-84: the header pass dispatches on the name, and an 

3558 # unnamed buffer read its binary member as CSV. 

3559 buf.name = member 

3560 member_plan = plan 

3561 if plan is not None and plan.columns: 

3562 member_plan = plan.narrowed_to(read_table_columns(buf)) 

3563 buf.seek(0) 

3564 frames.append(_read_by_extension(buf, name, member_plan)) 

3565 else: 

3566 # PERF-6: one archive can hold members with different 

3567 # columns, and the plan is built from their union — narrow 

3568 # it to this member's own header or `usecols` rejects it. 

3569 header, sep = _member_layout(zf, info) 

3570 member_plan = plan 

3571 if plan is not None and plan.columns: 

3572 member_plan = plan.narrowed_to(header) 

3573 try: 

3574 frames.append( 

3575 _read_by_extension( 

3576 io.BufferedReader(stream), name, member_plan, sep=sep 

3577 ) 

3578 ) 

3579 except UnicodeDecodeError: 

3580 # BUG-55: not UTF-8, and a member stream cannot be 

3581 # rewound for the encoding fallback — read it again 

3582 # into memory, under the same budget, where it can. 

3583 with zf.open(info) as again: 

3584 stream = _BudgetedZipMember( 

3585 again, member_budget, member, limit_label=limit_label 

3586 ) 

3587 buf = io.BytesIO(stream.read()) 

3588 frames.append( 

3589 _read_by_extension(buf, name, member_plan, sep=sep) 

3590 ) 

3591 remaining -= stream.consumed 

3592 labels.append(member) 

3593 return _tag_and_concat(frames, source_labels(labels), SOURCE_FILE_COLUMN) 

3594 

3595 

3596# BUG-5: guard the memory-constrained hosted demo against a too-large upload. 

3597# The bytes upload fine; the OOM comes later, when a big (often zipped) table 

3598# decompresses and pandas holds several copies through parse + normalization — 

3599# on Streamlit Community Cloud (~1 GB RAM) that silently kills the process with 

3600# no traceback. Above this raw-upload size the wizard warns and asks for an 

3601# explicit opt-in before parsing (a one-click confirm locally, real protection on 

3602# the host). Tuned to sit below the OneStop repeated-reading export 

3603# (~29–37 MB per zipped table) that first surfaced this. 

3604UPLOAD_SIZE_WARN_BYTES = 25 * 1024 * 1024 

3605 

3606 

3607def uploaded_files_total_bytes(uploaded) -> int: 

3608 """Total byte size of a Streamlit upload — one ``UploadedFile`` or a list. 

3609 

3610 Reads the ``.size`` each file already carries (no data copy). ``None`` / 

3611 empty → 0, so an absent upload is trivially under any threshold.""" 

3612 if not uploaded: 

3613 return 0 

3614 files = uploaded if isinstance(uploaded, (list, tuple)) else [uploaded] 

3615 return sum(int(getattr(f, "size", 0) or 0) for f in files) 

3616 

3617 

3618def upload_exceeds_limit( 

3619 uploaded, threshold_bytes: int = UPLOAD_SIZE_WARN_BYTES 

3620) -> bool: 

3621 """Whether a Streamlit upload's total size is over the guard threshold (BUG-5).""" 

3622 return uploaded_files_total_bytes(uploaded) > threshold_bytes 

3623 

3624 

3625def read_table( 

3626 file_like_or_path, *, plan: ReadPlan | None = None, kind: str | None = None 

3627) -> pd.DataFrame: 

3628 """Read a tabular file by extension: csv, tsv, parquet, feather, or a 

3629 ``.zip`` wrapping one or more of those (e.g. ``data.csv.zip``). A 

3630 multi-member zip is concatenated like a multi-file upload. 

3631 

3632 ``plan`` (PERF-6) is a :class:`ReadPlan` from :func:`plan_table_read`, 

3633 narrowing the read to the columns normalization keeps. ``kind`` 

3634 (``"words"`` / ``"fixations"``) is the table being read: a zip that mixes 

3635 the two keeps only the members that fit it (:func:`zip_member_split`).""" 

3636 name = getattr(file_like_or_path, "name", str(file_like_or_path)).lower() 

3637 try: 

3638 if name.endswith(".zip"): 

3639 return _read_zipped_table(file_like_or_path, plan=plan, kind=kind) 

3640 return _read_by_extension(file_like_or_path, name, plan) 

3641 except pd.errors.EmptyDataError as exc: 

3642 raise _empty_file_error(name) from exc 

3643 

3644 

3645def expand_table_inputs(inputs: TablesInput) -> list: 

3646 """Flatten a path / glob pattern / file-like / list-of-those into a list. 

3647 

3648 Glob patterns are expanded in sorted order so multi-file datasets (one 

3649 file per participant or per stimulus) can be referenced with a single 

3650 pattern like ``scanpaths/*.tsv``. Raises ``FileNotFoundError`` for a 

3651 pattern that matches nothing — silently loading zero files would read as 

3652 success.""" 

3653 if not isinstance(inputs, (list, tuple)): 

3654 inputs = [inputs] 

3655 expanded: list = [] 

3656 for item in inputs: 

3657 if isinstance(item, (str, os.PathLike)) and glob.has_magic(str(item)): 

3658 matches = sorted(glob.glob(str(item), recursive=True)) 

3659 if not matches: 

3660 raise FileNotFoundError(f"No files match pattern: {item}") 

3661 expanded.extend(matches) 

3662 else: 

3663 expanded.append(item) 

3664 return expanded 

3665 

3666 

3667def read_tables( 

3668 inputs: TablesInput, 

3669 source_column: str | None = SOURCE_FILE_COLUMN, 

3670 *, 

3671 plan_for=None, 

3672 kind: str | None = None, 

3673) -> pd.DataFrame: 

3674 """Read one or many tabular files and concatenate them into one frame. 

3675 

3676 ``inputs`` may be a single path or file-like object, a glob pattern, or a 

3677 list mixing those (a ``.zip`` member counts as a file too). ``plan_for`` is 

3678 called with each file's column names and returns the :class:`ReadPlan` to 

3679 read it under (PERF-6); omit it to parse every column. Each part gets a 

3680 ``source_file`` column holding the file's stem — qualified by its folders 

3681 when two files share one (:func:`source_labels`) — (unless the data already has 

3682 that column, or ``source_column=None``) — *including a single file*, so 

3683 datasets that key identity in the filename can recover it (the upload wizard 

3684 maps ``source_file`` as the trial / participant id). Columns are aligned by 

3685 name across files; fields absent from a file become NaN for its rows. 

3686 ``kind`` picks a mixed zip's members (:func:`zip_member_split`).""" 

3687 items = expand_table_inputs(inputs) 

3688 frames, labels = [], [] 

3689 for item in items: 

3690 # PERF-6: each file is planned against its OWN header. One file per 

3691 # participant is the common upload shape, and an export can gain or 

3692 # lose a column between them — a shared plan would name a column some 

3693 # file hasn't got, which `usecols` raises on. 

3694 plan = ( 

3695 plan_for(read_table_columns(item, kind=kind)) 

3696 if plan_for is not None 

3697 else None 

3698 ) 

3699 frames.append(read_table(item, plan=plan, kind=kind)) 

3700 labels.append(getattr(item, "name", str(item))) 

3701 return _tag_and_concat( 

3702 frames, source_labels(labels), source_column, always_tag=True 

3703 ) 

3704 

3705 

3706def _load_bundled(name: str) -> pd.DataFrame: 

3707 """Load a single bundled sample, preferring Parquet over CSV.""" 

3708 data_root = resources.files(PACKAGE_NAME).joinpath("sample_data") 

3709 for ext in (".parquet", ".csv"): 

3710 resource = data_root / f"{name}{ext}" 

3711 try: 

3712 with resources.as_file(resource) as path: 

3713 if not path.is_file(): 

3714 continue 

3715 return read_table(path) 

3716 except FileNotFoundError: 

3717 continue 

3718 return pd.DataFrame() 

3719 

3720 

3721def _resolve_sample_image_paths(df: pd.DataFrame) -> pd.DataFrame: 

3722 """Expand the bundled demo's relative ``image_path`` (e.g. 

3723 ``images/2_2_1_Adv__paragraph.png``) into an absolute path under the 

3724 packaged ``sample_data`` directory, so the stimulus-image background layer 

3725 (``tabs._render_single_trial`` → ``plots.make_scanpath_figure``) can load it 

3726 via ``os.path.exists``. The CSVs ship a *relative* reference (stable across 

3727 installs); this resolves it at load time against wherever the wheel landed. 

3728 No-op when the column is absent or a value is already absolute.""" 

3729 if "image_path" not in df.columns: 

3730 return df 

3731 try: 

3732 root = Path(str(resources.files(PACKAGE_NAME).joinpath("sample_data"))) 

3733 except (ModuleNotFoundError, FileNotFoundError, TypeError): 

3734 return df 

3735 

3736 def _abs(value: object) -> object: 

3737 if isinstance(value, str) and value and not os.path.isabs(value): 

3738 return str(root.joinpath(*value.split("/"))) 

3739 return value 

3740 

3741 df = df.copy() 

3742 df["image_path"] = df["image_path"].map(_abs) 

3743 return df 

3744 

3745 

3746def _pattern_placeholders(pattern: str) -> list[str]: 

3747 """The row fields a filename pattern reads, in first-appearance order. 

3748 

3749 ``"{text_id}/{trial_id:0>3}.png"`` reads ``text_id`` and ``trial_id``. 

3750 An attribute or index suffix is trimmed back to the field the row is keyed 

3751 by (``{a.b}`` reads ``a``), matching the row-dict lookup this feeds. 

3752 Auto-numbered fields (``{}``) name no row field and are skipped — 

3753 ``format_map`` raises for them whether or not they are collected. 

3754 """ 

3755 names: list[str] = [] 

3756 for _, field_name, _, _ in string.Formatter().parse(pattern): 

3757 if not field_name: 

3758 continue 

3759 name = re.split(r"[.\[]", field_name, maxsplit=1)[0] 

3760 if name and name not in names: 

3761 names.append(name) 

3762 return names 

3763 

3764 

3765def resolve_stimulus_image_paths( 

3766 frame: pd.DataFrame, 

3767 root: str | os.PathLike, 

3768 pattern: str = "{text_id}.png", 

3769 *, 

3770 require_exists: bool = True, 

3771) -> pd.DataFrame: 

3772 """Attach per-row stimulus images from a local folder and filename pattern. 

3773 

3774 Placeholders are read from the row (for example ``{text_id}``, 

3775 ``{trial_id}``, or ``{participant_id}``). Relative subdirectories are 

3776 supported, but resolved files must remain under ``root``; absolute and 

3777 parent-traversal patterns are rejected. Rows whose placeholders are missing 

3778 or whose file does not exist keep their previous ``image_path`` value. 

3779 

3780 This pure helper is the headless/API surface for VIZ-14 and is also used by 

3781 the desktop-only folder controls and the CLI. It is deliberately 

3782 **undecorated**: the CLI and the API have no Streamlit session to cache 

3783 into, so any caching belongs at a call site rather than here. 

3784 

3785 **One filesystem probe per distinct placeholder tuple, not per row.** A 

3786 pattern reads a couple of columns, so a fixation frame of millions of rows 

3787 still resolves only a few hundred distinct paths — where the per-row form 

3788 this replaced spent two syscalls each (``Path.resolve`` + ``is_file``) on 

3789 *every* rerun that had a folder attached. Per-row behaviour is unchanged: 

3790 the fallback is each row's **own** prior ``image_path``, not a shared one, 

3791 and a NaN placeholder is still absent from the format mapping rather than 

3792 the string ``"nan"`` (which would otherwise resolve ``nan.png`` — a miss 

3793 today, a wrong hit the day such a file exists). 

3794 """ 

3795 if frame is None or frame.empty: 

3796 return frame.copy() if isinstance(frame, pd.DataFrame) else pd.DataFrame() 

3797 base = Path(root).expanduser().resolve() 

3798 if not pattern or Path(pattern).is_absolute(): 

3799 raise ValueError("Image filename pattern must be a non-empty relative path.") 

3800 

3801 class _Row(dict): 

3802 def __missing__(self, key): 

3803 raise KeyError(key) 

3804 

3805 def _resolved(values: dict[str, str]) -> str | None: 

3806 """One placeholder tuple to an absolute path, or None to keep `previous`.""" 

3807 try: 

3808 relative = pattern.format_map(_Row(values)) 

3809 except (KeyError, ValueError, AttributeError): 

3810 return None 

3811 candidate = (base / relative).resolve() 

3812 try: 

3813 candidate.relative_to(base) 

3814 except ValueError as exc: 

3815 raise ValueError( 

3816 f"Image pattern resolves outside the selected folder: {relative!r}" 

3817 ) from exc 

3818 if require_exists and not candidate.is_file(): 

3819 return None 

3820 return str(candidate) 

3821 

3822 # Keyed by the *stringified* column label, as the per-row dict was; a 

3823 # duplicate label keeps the last column, as that dict's later write did. 

3824 by_name: dict[str, object] = {str(column): column for column in frame.columns} 

3825 used = [by_name[name] for name in _pattern_placeholders(pattern) if name in by_name] 

3826 

3827 if used: 

3828 text = pd.DataFrame(index=frame.index) 

3829 missing = np.zeros(len(frame), dtype=bool) 

3830 for position, column in enumerate(used): 

3831 values = frame[column] 

3832 if isinstance(values, pd.DataFrame): # duplicate column labels 

3833 values = values.iloc[:, -1] 

3834 missing |= values.isna().to_numpy() 

3835 text[position] = values.astype(str) 

3836 codes, uniques = pd.factorize(pd.MultiIndex.from_frame(text)) 

3837 names = [str(column) for column in used] 

3838 resolved = np.asarray( 

3839 [_resolved(dict(zip(names, key))) for key in uniques], dtype=object 

3840 )[codes] 

3841 # A NaN placeholder never reached the format mapping, so the row keeps 

3842 # its own `image_path`; `astype(str)` above turned it into "nan". 

3843 resolved[missing] = None 

3844 else: 

3845 resolved = np.full(len(frame), _resolved({}), dtype=object) 

3846 

3847 series = pd.Series(resolved, index=frame.index, dtype=object) 

3848 if "image_path" in frame.columns: 

3849 series = series.where(series.notna(), frame["image_path"]) 

3850 result = frame.copy() 

3851 result["image_path"] = series 

3852 return result 

3853 

3854 

3855@st.cache_data 

3856def load_sample_data() -> tuple[pd.DataFrame, pd.DataFrame]: 

3857 """Load bundled demo IA and fixation tables (prefer Parquet). 

3858 

3859 The tables ship per-trial stimulus-image references (``image_path`` + 

3860 ``image_x``/``image_y`` origin) pointing at the rendered paragraph PNGs 

3861 under ``sample_data/images/``; the relative paths are resolved to absolute 

3862 here so the optional stimulus-image background layer renders the page 

3863 behind the scanpath (aligned to the 2560x1440 OneStop monitor).""" 

3864 words = _resolve_sample_image_paths(_load_bundled("ia")) 

3865 fixations = _resolve_sample_image_paths(_load_bundled("fixations")) 

3866 if words.empty or fixations.empty: 

3867 st.error( 

3868 "The bundled demo is missing from this installation. Reinstall " 

3869 "Scanpath Studio, then reload the page." 

3870 ) 

3871 return pd.DataFrame(), pd.DataFrame() 

3872 return words, fixations 

3873 

3874 

3875@st.cache_data 

3876def load_sample_raw_gaze() -> pd.DataFrame: 

3877 """Load bundled raw gaze sample (millisecond-level x,y).""" 

3878 return _load_bundled("raw_gaze") 

3879 

3880 

3881def infer_raw_gaze_schema(raw_gaze: pd.DataFrame) -> dict[str, str] | None: 

3882 """Infer schema for raw millisecond-level gaze data.""" 

3883 schema = propose_raw_gaze_schema(raw_gaze) 

3884 problems = validate_raw_gaze_schema(schema) 

3885 if problems: 

3886 st.error(f"Missing required raw gaze fields: {', '.join(problems)}") 

3887 return None 

3888 return schema 

3889 

3890 

3891def normalize_raw_gaze( 

3892 raw_gaze: pd.DataFrame, schema: dict[str, str], *, keep_columns: set | None = None 

3893) -> pd.DataFrame: 

3894 """Normalize raw gaze data to canonical column names. 

3895 

3896 ``keep_columns`` (UX-120) carries user-chosen extra source columns 

3897 through verbatim, the same "Extra fields to keep" mechanism 

3898 :func:`normalize_words`/:func:`normalize_fixations` already have — raw 

3899 gaze has no optional-fields *registry* of its own (no semantic field 

3900 here is common enough across exports to earn a canonical name the way 

3901 ``saccade_amplitude`` does for fixations), so this only ever carries 

3902 unclaimed columns, via :func:`_carry_extra_columns`. 

3903 """ 

3904 # Samples with no participant or trial id belong to no trial (BUG-56's 

3905 # rule for the other two tables, round 10). 

3906 blank, unkeyed = _rows_missing_identity(raw_gaze, schema) 

3907 for issue in identity_issues(raw_gaze, schema, table="Raw gaze", unkeyed=unkeyed): 

3908 warnings.warn(issue.replace("`", "'"), UserWarning, stacklevel=2) 

3909 if (blank | unkeyed).any(): 

3910 raw_gaze = raw_gaze.loc[~(blank | unkeyed)] 

3911 df = pd.DataFrame(index=raw_gaze.index) 

3912 if schema.get("participant"): 

3913 # str or list (a composite participant id), joined like normalize_words — 

3914 # so a composite participant stays key-compatible with the fixations. 

3915 df["participant_id"] = trial_id_series(raw_gaze, schema["participant"]) 

3916 else: 

3917 df["participant_id"] = SYNTHETIC_PARTICIPANT 

3918 trial_cols = trial_mapping_columns(schema["trial"]) 

3919 if len(trial_cols) > 1: 

3920 # User-composed unique trial ID — see normalize_words. 

3921 df["trial_id"] = trial_id_series(raw_gaze, trial_cols) 

3922 df["unique_trial_id"] = df["trial_id"] 

3923 else: 

3924 trial_col = trial_cols[0] # the mapped column (BUG-58) 

3925 df["trial_id"] = stable_id(raw_gaze[trial_col]) 

3926 if "unique_trial_id" in raw_gaze.columns: 

3927 # The mapped id *is* the unique trial id (BUG-58) — never the raw 

3928 # column's own values, which the trial picker would otherwise key 

3929 # on (`utils.build_combo_options` prefers `unique_trial_id`). 

3930 df["unique_trial_id"] = df["trial_id"] 

3931 # UX-113: mapped when the export carries its own text/passage column; 

3932 # otherwise raw gaze has no text/passage concept of its own, so mirror 

3933 # trial_id — a raw-gaze-only dataset still needs *a* text_id column for 

3934 # the trial picker (utils.build_combo_options). 

3935 if schema.get("text_id"): 

3936 df["text_id"] = raw_gaze[schema["text_id"]].astype(str) 

3937 else: 

3938 df["text_id"] = df["trial_id"] 

3939 df = _copy_screen_fields(df, raw_gaze, schema) 

3940 if schema.get("text"): 

3941 df["text"] = raw_gaze[schema["text"]].astype(str) 

3942 else: 

3943 df["text"] = "" 

3944 # UX-113: raw gaze samples are never run through word assignment the way 

3945 # fixations are (there is no per-sample geometry step for it) — carried 

3946 # through only when the export already names one, not computed. 

3947 if schema.get("word_id"): 

3948 df["word_id"] = raw_gaze[schema["word_id"]] 

3949 df["x"] = _to_number(raw_gaze[schema["x"]]) 

3950 df["y"] = _to_number(raw_gaze[schema["y"]]) 

3951 skip = _schema_source_columns(schema) 

3952 if schema.get("timestamp"): 

3953 onset = schema["timestamp"] 

3954 df["timestamp_ms"] = _as_ms(_to_number(raw_gaze[onset]), onset) # DATA-40 

3955 else: 

3956 # No clock: the samples keep their order (1, 2, … per trial) and no 

3957 # time. Nothing says how far apart they were recorded, so a made-up 

3958 # `timestamp_ms` would be a sampling rate the data never stated — and 

3959 # it would reach the plot's colour scale, hover and the exports as ms. 

3960 df[SAMPLE_INDEX] = df.groupby(list(PARENT_KEY), sort=False).cumcount() + 1 

3961 # A kept extra named `timestamp_ms` would be read as that clock again. 

3962 skip = skip | {"timestamp_ms"} 

3963 if keep_columns is not None: 

3964 _carry_extra_columns(df, raw_gaze, keep_columns, skip) 

3965 df = _preserve_composite_columns(df, raw_gaze, schema["trial"]) 

3966 return df 

3967 

3968 

3969def infer_word_schema(words: pd.DataFrame) -> dict[str, str] | None: 

3970 schema = propose_word_schema(words) 

3971 problems = validate_word_schema(schema) 

3972 if problems: 

3973 st.error(f"Words table problems: {'; '.join(problems)}") 

3974 return None 

3975 return schema 

3976 

3977 

3978def infer_fix_schema(fixations: pd.DataFrame) -> dict[str, str] | None: 

3979 schema = propose_fix_schema(fixations) 

3980 problems = validate_fix_schema(schema) 

3981 if problems: 

3982 st.error(f"Fixations schema problems: {'; '.join(problems)}") 

3983 return None 

3984 return schema 

3985 

3986 

3987# Placeholder participant id for stimulus-level word/AoI tables (no participant 

3988# column — one row per word per text, shared by every reading). The marker 

3989# column flags the frame so broadcast_stimulus_words() knows to expand it. 

3990STIMULUS_PARTICIPANT = "" 

3991STIMULUS_WORDS_FLAG = "_stimulus_words" 

3992 

3993# Synthetic participant id used when a dataset has no participant column at all 

3994# (a single anonymous reader). Distinct from STIMULUS_PARTICIPANT ("") so it 

3995# never collides with the stimulus-word broadcast machinery — participant_id is 

3996# always present downstream (combos/filters/annotations/export/measures groupby), 

3997# and the UI hides the participant selector when there's only this one value. 

3998SYNTHETIC_PARTICIPANT = "(all)" 

3999 

4000#: The two keys a stimulus-level AOI table can attach to a reading through 

4001#: (DATA-49), as the wizard names them. 

4002STIMULUS_JOIN_LABELS = {"trial_id": "Trial ID", "text_id": "Text ID"} 

4003#: The mapped trial id a repeated reading had before 

4004#: `_disambiguate_repeated_readings` suffixed it with `_r2`, `_r3` … — on the 

4005#: fixations only, and only when a suffix was given. The stimulus join reads it 

4006#: so a repeat finds the boxes of what it re-read (BUG-57, DATA-49). 

4007BASE_TRIAL_ID = "_base_trial_id" 

4008#: On a broadcast words copy: the AOI table's own trial id it was copied from 

4009#: (the reading supplies `trial_id`). ✏️ Edit dataset keeps one copy per value 

4010#: of it to get the stimulus table back (`remap_normalized_frame`). 

4011AOI_TRIAL_ID = "_aoi_trial_id" 

4012#: Whether this frame's `text_id` came from a mapped Text ID (a schema pick, 

4013#: auto-detected or not, or a literal `unique_paragraph_id`) rather than the 

4014#: trial-id fallback — written by `normalize_*` from the schema, so a mapped 

4015#: Text ID whose values happen to equal the trial ids still counts (DATA-49). 

4016TEXT_ID_MAPPED = "_text_id_mapped" 

4017#: On a fixation: its `timestamp_ms` was made up by `normalize_fixations` 

4018#: because the table mapped no onset — the reading order 0, 1, 2, …, kept so 

4019#: fixations still sort, but not a time. Anything that needs elapsed time 

4020#: (the summaries' reading time and speed, the replay clock) lays the 

4021#: fixations end to end by their durations instead and says it is an estimate. 

4022TIMESTAMP_SYNTHESIZED = "_timestamp_synthesized" 

4023#: Bookkeeping columns the pipeline needs and the user never sees: kept in the 

4024#: frames and the recovery cache, dropped from exports and the Data page's 

4025#: tables (`drop_internal_columns`), and never offered as a field — their 

4026#: leading underscore is what the field listers skip. 

4027INTERNAL_COLUMNS = frozenset( 

4028 { 

4029 STIMULUS_WORDS_FLAG, 

4030 BASE_TRIAL_ID, 

4031 AOI_TRIAL_ID, 

4032 TEXT_ID_MAPPED, 

4033 TIMESTAMP_SYNTHESIZED, 

4034 } 

4035) 

4036 

4037 

4038def timestamps_synthesized(fixations: pd.DataFrame | None) -> bool: 

4039 """Whether any of these fixations has a made-up ``timestamp_ms``. 

4040 

4041 True when normalization had no onset column to read and numbered the 

4042 fixations instead (:data:`TIMESTAMP_SYNTHESIZED`). A frame without the 

4043 column — one built by hand, or stored before it existed — counts as 

4044 recorded, which is what it always did.""" 

4045 if fixations is None or TIMESTAMP_SYNTHESIZED not in fixations.columns: 

4046 return False 

4047 return bool(fixations[TIMESTAMP_SYNTHESIZED].fillna(False).astype(bool).any()) 

4048 

4049 

4050#: Scratch column the stimulus broadcast merges through. 

4051_STIMULUS_KEY = "_stimulus_key" 

4052 

4053 

4054def user_columns(frame: pd.DataFrame) -> list: 

4055 """``frame``'s columns a user can pick — every one but :data:`INTERNAL_COLUMNS`. 

4056 

4057 The one list every widget offering a frame's fields should draw from, so a 

4058 bookkeeping column can never surface as a hover field, a Q&A field, a 

4059 comparison field or a mapping option (DATA-49).""" 

4060 return [c for c in frame.columns if c not in INTERNAL_COLUMNS] 

4061 

4062 

4063def drop_internal_columns(frame: pd.DataFrame) -> pd.DataFrame: 

4064 """``frame`` without :data:`INTERNAL_COLUMNS` (the same object if none).""" 

4065 present = [c for c in INTERNAL_COLUMNS if c in frame.columns] 

4066 return frame.drop(columns=present) if present else frame 

4067 

4068 

4069def shareable_frame(frame: pd.DataFrame) -> pd.DataFrame: 

4070 """``frame`` as it leaves the app: no bookkeeping, no made-up clock. 

4071 

4072 :func:`drop_internal_columns`, and also ``timestamp_ms`` when normalization 

4073 numbered the fixations itself (:func:`timestamps_synthesized`): those 

4074 numbers are a sort order, and a file that carried them under that name 

4075 would read back as recorded milliseconds.""" 

4076 if timestamps_synthesized(frame) and "timestamp_ms" in frame.columns: 

4077 frame = frame.drop(columns="timestamp_ms") 

4078 return drop_internal_columns(frame) 

4079 

4080 

4081class StimulusJoinError(ValueError): 

4082 """A stimulus-level AOI table that the readings in the fixations can't use. 

4083 

4084 Raised by :func:`broadcast_stimulus_words` — so by ``harmonize_frames`` and 

4085 every loader built on it — instead of returning a dataset with no word 

4086 boxes (or, on a multipart dataset, with screens that have none): the 

4087 add-dataset wizard and ✏️ Edit dataset block on it, the headless API raises 

4088 it, and the CLI prints it (DATA-49).""" 

4089 

4090 

4091class StimulusJoinWarning(UserWarning): 

4092 """Some readings found no stimulus-level word boxes (DATA-49); the rest did. 

4093 

4094 A ``UserWarning``, like the normalizers' load warnings, so the API and the 

4095 CLI see it; its own class so ✏️ Edit dataset can catch it during a save and 

4096 show it on the page.""" 

4097 

4098 

4099def _count(n: int, singular: str, plural: str) -> str: 

4100 return f"{n:,} {singular if n == 1 else plural}" 

4101 

4102 

4103@dataclass(frozen=True) 

4104class StimulusJoin: 

4105 """How a stimulus-level AOI table attached to the readings (DATA-49). 

4106 

4107 A *reading* is one ``(participant_id, trial_id)`` of the fixations — one 

4108 ``(participant_id, trial_id, screen_id)`` on a multipart dataset, where the 

4109 counts are of *reading screens*. Each takes the boxes of the AOI table's 

4110 trial it is matched to by the first of these that finds one: its own trial 

4111 id, the trial id it had before a repeat's ``_r2`` suffix 

4112 (``_base_trial_id``), or its Text ID. ``by_trial`` were matched the first 

4113 two ways, ``by_text`` the third. A trial-id match always stands; when both 

4114 tables map a real Text ID and the reading's disagrees with the matched AOI 

4115 trial's, it is counted in ``text_mismatches`` (``mismatch_example`` = 

4116 ``(reading's, AOI trial's)``) and warned about, never redirected. 

4117 ``ambiguous_texts`` are Text IDs the AOI table gives to more than one of its 

4118 trials; ``ambiguous_readings`` of the unmatched readings name one of them, 

4119 and ``missing_screen_readings`` match an AOI trial that has no boxes for 

4120 their screen (``missing_screens``, examples).""" 

4121 

4122 readings: int 

4123 by_trial: int 

4124 by_text: int 

4125 ambiguous_texts: tuple[str, ...] = () 

4126 ambiguous_readings: int = 0 

4127 missing_screen_readings: int = 0 

4128 missing_screens: tuple[str, ...] = () 

4129 text_mismatches: int = 0 

4130 mismatch_example: tuple[str, str] | None = None 

4131 multipart: bool = False 

4132 

4133 @property 

4134 def matched(self) -> int: 

4135 """Readings (reading screens) that got word boxes.""" 

4136 return self.by_trial + self.by_text 

4137 

4138 @property 

4139 def unmatched(self) -> int: 

4140 return self.readings - self.matched 

4141 

4142 @property 

4143 def needs_warning(self) -> bool: 

4144 """Whether :meth:`describe` reports something to act on: readings with 

4145 no boxes, or readings whose Text IDs disagree with their boxes'.""" 

4146 return bool(self.unmatched or self.text_mismatches) 

4147 

4148 @property 

4149 def key(self) -> str | None: 

4150 """``"trial_id"`` / ``"text_id"`` when every matched reading took the 

4151 same route, ``"mixed"`` when both were used, ``None`` for no match.""" 

4152 if not self.matched: 

4153 return None 

4154 if not self.by_text: 

4155 return "trial_id" 

4156 return "text_id" if not self.by_trial else "mixed" 

4157 

4158 @property 

4159 def label(self) -> str: 

4160 """The route as the wizard's fields name it (``"Text ID"``).""" 

4161 if self.key == "mixed": 

4162 return f"Trial ID ({self.by_trial:,}) and by Text ID ({self.by_text:,})" 

4163 return STIMULUS_JOIN_LABELS.get(self.key or "", "") 

4164 

4165 def _unit(self, n: int) -> str: 

4166 return ( 

4167 _count(n, "trial screen", "trial screens") 

4168 if self.multipart 

4169 else _count(n, "trial", "trials") 

4170 ) 

4171 

4172 def _why_unmatched(self) -> str: 

4173 """Why the unmatched readings found nothing, split by the reason.""" 

4174 parts = [] 

4175 plain = self.unmatched - self.ambiguous_readings - self.missing_screen_readings 

4176 if plain: 

4177 verb = "shares" if plain == 1 else "share" 

4178 parts.append( 

4179 f" {self._unit(plain)} {verb} neither a Trial ID nor a Text ID " 

4180 "with the Words table." 

4181 ) 

4182 if self.missing_screen_readings: 

4183 screens = ", ".join(repr(s) for s in self.missing_screens[:3]) 

4184 verb = "matches" if self.missing_screen_readings == 1 else "match" 

4185 which = "that screen" if len(self.missing_screens) == 1 else "those screens" 

4186 parts.append( 

4187 f" {self._unit(self.missing_screen_readings)} {verb} a Words-table trial " 

4188 f"that has no boxes for {which} ({screens})." 

4189 ) 

4190 if self.ambiguous_readings: 

4191 shown = ", ".join(repr(t) for t in self.ambiguous_texts[:3]) 

4192 more = len(self.ambiguous_texts) - 3 

4193 if more > 0: 

4194 shown += f" and {more:,} more" 

4195 texts = ( 

4196 "that Text ID" if len(self.ambiguous_texts) == 1 else "those Text IDs" 

4197 ) 

4198 verb = "names" if self.ambiguous_readings == 1 else "name" 

4199 parts.append( 

4200 f" {self._unit(self.ambiguous_readings)} {verb} a Text ID the Words " 

4201 f"table gives to more than one of its trials ({shown}), so " 

4202 f"{texts} cannot pick one set of boxes." 

4203 ) 

4204 return "".join(parts) 

4205 

4206 def _mismatch_note(self) -> str: 

4207 if not self.text_mismatches: 

4208 return "" 

4209 verb, pron = ("has", "its") if self.text_mismatches == 1 else ("have", "their") 

4210 example = "" 

4211 if self.mismatch_example is not None: 

4212 reading, aoi = self.mismatch_example 

4213 example = f" (e.g. {reading!r} against {aoi!r})" 

4214 return ( 

4215 f" {self._unit(self.text_mismatches)} {verb} the boxes of the Words-" 

4216 f"table trial {pron} Trial ID matches, but a different Text ID from that " 

4217 f"trial's{example}: check that Text ID names the same texts in both " 

4218 "tables." 

4219 ) 

4220 

4221 def describe(self) -> str: 

4222 """One sentence for the wizard and the log: the route and its coverage.""" 

4223 if self.key is None or (self.multipart and self.unmatched): 

4224 return self.problem() 

4225 unit = "trial screens" if self.multipart else "trials" 

4226 lead = f"Words attach to {unit} by {self.label}" 

4227 if not self.unmatched: 

4228 return ( 

4229 f"{lead}: all {self.readings:,} {unit} have word boxes." 

4230 f"{self._mismatch_note()}" 

4231 ) 

4232 return ( 

4233 f"{lead}: {self.matched:,} of {self._unit(self.readings)} have word " 

4234 f"boxes.{self._why_unmatched()}{self._mismatch_note()}" 

4235 ) 

4236 

4237 def problem(self) -> str: 

4238 """Why the join is refused, and what to map instead.""" 

4239 advice = ( 

4240 " Map **Text ID** in both tables to the column naming the text " 

4241 "each row belongs to, or give the Words table the fixations' own " 

4242 "Trial IDs." 

4243 ) 

4244 if self.matched: 

4245 # Only a multipart dataset refuses a partial join: every screen a 

4246 # reading has fixations on needs its boxes (`validate_matching_parts`). 

4247 return ( 

4248 "The Words table has no Participant ID, so its word boxes are shared " 

4249 f"by every trial of a text, but only {self.matched:,} of " 

4250 f"{self._unit(self.readings)} find theirs, and a multipart dataset " 

4251 "needs boxes for every screen it has fixations on." 

4252 f"{self._why_unmatched()}{advice}" 

4253 ) 

4254 return ( 

4255 "The Words table has no Participant ID, so its word boxes are shared by " 

4256 f"every trial of a text, but none of the {self._unit(self.readings)} " 

4257 "in the fixations finds them: the dataset would have no word boxes." 

4258 f"{self._why_unmatched()}{advice}" 

4259 ) 

4260 

4261 

4262def _as_key(values: pd.Series) -> pd.Series: 

4263 """An id column as the strings the join compares, missing ids kept missing 

4264 (``astype(str)`` alone would spell them ``"nan"`` and match each other).""" 

4265 return values.astype(str).where(values.notna()) 

4266 

4267 

4268def _text_is_fallback(frame: pd.DataFrame) -> bool: 

4269 """Whether ``text_id`` is only the trial-id fallback on this frame. 

4270 

4271 Read from ``_text_id_mapped``, which ``normalize_*`` write from the schema, 

4272 so a mapped Text ID is mapped even when its values equal the trial ids. 

4273 Only a frame normalized without it (a stored dataset from before, frames 

4274 built by hand) falls back to the values: with no Text ID mapped, 

4275 ``normalize_*`` copy the (unsuffixed) trial id into ``text_id``, so a 

4276 ``text_id`` equal to that id on every row says nothing about texts. 

4277 Compared over distinct pairs, so it is cheap on a million rows.""" 

4278 if TEXT_ID_MAPPED in frame.columns: 

4279 return not bool(frame[TEXT_ID_MAPPED].fillna(False).astype(bool).any()) 

4280 columns = ["text_id", "trial_id"] + [BASE_TRIAL_ID] * (BASE_TRIAL_ID in frame) 

4281 pairs = frame[columns].drop_duplicates() 

4282 trial = pairs["trial_id"] 

4283 if BASE_TRIAL_ID in pairs: 

4284 trial = pairs[BASE_TRIAL_ID].where(pairs[BASE_TRIAL_ID].notna(), trial) 

4285 return bool( 

4286 ( 

4287 _as_key(pairs["text_id"]).fillna("\x00") == _as_key(trial).fillna("\x00") 

4288 ).all() 

4289 ) 

4290 

4291 

4292def _lookup( 

4293 readings: pd.DataFrame, 

4294 keys: pd.Series, 

4295 table: pd.DataFrame, 

4296 screen: list, 

4297 value: str = _STIMULUS_KEY, 

4298) -> np.ndarray: 

4299 """For each reading, ``table[value]`` where ``table["_key"]`` equals ``keys``. 

4300 

4301 ``table`` is unique on ``_key`` + ``screen``; a left merge keeps the 

4302 readings' order, so the result lines up with them (missing = no match). 

4303 A blank screen id pairs with a blank one, as the merge always has.""" 

4304 probe = pd.DataFrame( 

4305 {"_key": keys.to_numpy(), **{c: readings[c].to_numpy() for c in screen}} 

4306 ) 

4307 hits = probe.merge( 

4308 table[["_key", *screen, value]], on=["_key", *screen], how="left" 

4309 ) 

4310 return hits[value].to_numpy() 

4311 

4312 

4313def _plan_stimulus_join( 

4314 words: pd.DataFrame, fixations: pd.DataFrame 

4315) -> tuple[StimulusJoin, pd.DataFrame, list[str]]: 

4316 """The join, the fixations' distinct readings, and the reading key. 

4317 

4318 Each reading's ``_STIMULUS_KEY`` is the AOI table trial whose boxes it 

4319 takes (missing when none): its exact trial id, else its unsuffixed one, 

4320 else — only when the fixations map a real Text ID — its Text ID's one 

4321 trial. Vectorised over *distinct* keys (one ``drop_duplicates`` per frame 

4322 and one merge per route), so it costs the same on a million-row corpus as 

4323 the broadcast.""" 

4324 screen = [SCREEN_ID] * ( 

4325 SCREEN_ID in words.columns and SCREEN_ID in fixations.columns 

4326 ) 

4327 reading_key = ["participant_id", "trial_id", *screen] 

4328 has_text = "text_id" in words.columns and "text_id" in fixations.columns 

4329 # A fixations Text ID that is only its own trial id names no text, so it 

4330 # can neither find a text's boxes nor contradict them; the AOI table's 

4331 # fallback text *is* its stimulus id, which a real reading Text ID can use. 

4332 reading_text_real = has_text and not _text_is_fallback(fixations) 

4333 aoi_text_real = has_text and not _text_is_fallback(words) 

4334 extra = ["text_id"] * has_text + [BASE_TRIAL_ID] * ( 

4335 BASE_TRIAL_ID in fixations.columns 

4336 ) 

4337 readings = fixations[reading_key + extra].drop_duplicates(reading_key) 

4338 readings = readings.reset_index(drop=True) 

4339 trials = words[["trial_id", *screen, *["text_id"] * has_text]] 

4340 trials = trials.dropna(subset=["trial_id"]).drop_duplicates(["trial_id", *screen]) 

4341 trials = trials.assign(_key=_as_key(trials["trial_id"])) 

4342 trials[_STIMULUS_KEY] = trials["_key"] 

4343 

4344 def by_trial_routes(table: pd.DataFrame, on_screen: list) -> np.ndarray: 

4345 found = _lookup(readings, _as_key(readings["trial_id"]), table, on_screen) 

4346 if BASE_TRIAL_ID in readings.columns: 

4347 # A repeat's `_r2` tells the readings apart; it never costs the 

4348 # repeat its boxes (BUG-57): try the id it was recorded under. 

4349 base = _lookup(readings, _as_key(readings[BASE_TRIAL_ID]), table, on_screen) 

4350 found = np.where(pd.isna(found), base, found) 

4351 return found 

4352 

4353 stimulus = by_trial_routes(trials, screen) 

4354 by_trial = int(pd.notna(stimulus).sum()) 

4355 

4356 ambiguous: tuple[str, ...] = () 

4357 ambiguous_readings = mismatches = 0 

4358 mismatch_example = None 

4359 texts = None 

4360 if has_text: 

4361 reading_text = _as_key(readings["text_id"]) 

4362 if reading_text_real and aoi_text_real: 

4363 # An exact trial-id match always stands (a Text ID mapped at another 

4364 # grain, or on one side only, must never move a reading's boxes); 

4365 # when both tables name real texts and they disagree, say so. 

4366 matched_text = _lookup( 

4367 readings, 

4368 pd.Series(stimulus), 

4369 trials.assign(_text=_as_key(trials["text_id"])), 

4370 screen, 

4371 value="_text", 

4372 ) 

4373 disagree = ( 

4374 pd.notna(stimulus) 

4375 & pd.notna(matched_text) 

4376 & reading_text.notna().to_numpy() 

4377 & (matched_text != reading_text.to_numpy()) 

4378 ) 

4379 mismatches = int(disagree.sum()) 

4380 if mismatches: 

4381 i = int(np.flatnonzero(disagree)[0]) 

4382 mismatch_example = (str(reading_text.iloc[i]), str(matched_text[i])) 

4383 if reading_text_real: 

4384 text_key = ["text_id", *screen] 

4385 # A Text ID stands in for the trial only where it names one set of 

4386 # boxes: one the AOI table gives to several of its trials (two 

4387 # versions of a paragraph, an article id over paragraph boxes) would 

4388 # hand a reading all of them. Only those texts are left out. 

4389 per_text = words[[*text_key, "trial_id"]] 

4390 per_text = per_text.dropna(subset=["text_id", "trial_id"]).drop_duplicates() 

4391 shared = per_text.duplicated(text_key, keep=False) 

4392 ambiguous = tuple( 

4393 sorted(per_text.loc[shared, "text_id"].astype(str).unique()) 

4394 ) 

4395 chosen = per_text[~shared] 

4396 texts = pd.DataFrame( 

4397 { 

4398 "_key": _as_key(chosen["text_id"]).to_numpy(), 

4399 **{c: chosen[c].to_numpy() for c in screen}, 

4400 _STIMULUS_KEY: _as_key(chosen["trial_id"]).to_numpy(), 

4401 } 

4402 ) 

4403 by_text_id = _lookup(readings, reading_text, texts, screen) 

4404 stimulus = np.where(pd.isna(stimulus), by_text_id, stimulus) 

4405 if ambiguous: 

4406 ambiguous_readings = int( 

4407 (pd.isna(stimulus) & reading_text.isin(ambiguous).to_numpy()).sum() 

4408 ) 

4409 matched = int(pd.notna(stimulus).sum()) 

4410 

4411 missing_screen_readings = 0 

4412 missing_screens: tuple[str, ...] = () 

4413 if screen and matched < len(readings): 

4414 # Matched an AOI trial, just not on this screen: say which screen. 

4415 anywhere = by_trial_routes(trials.drop_duplicates(["trial_id"]), []) 

4416 if texts is not None: 

4417 anywhere = np.where( 

4418 pd.isna(anywhere), 

4419 _lookup( 

4420 readings, 

4421 _as_key(readings["text_id"]), 

4422 texts.drop_duplicates(["_key"]), 

4423 [], 

4424 ), 

4425 anywhere, 

4426 ) 

4427 missing = pd.isna(stimulus) & pd.notna(anywhere) 

4428 missing_screen_readings = int(missing.sum()) 

4429 missing_screens = tuple( 

4430 dict.fromkeys(str(s) for s in readings.loc[missing, SCREEN_ID]) 

4431 ) 

4432 if ambiguous_readings: 

4433 ambiguous_readings = int( 

4434 ( 

4435 pd.isna(stimulus) 

4436 & ~missing 

4437 & _as_key(readings["text_id"]).isin(ambiguous).to_numpy() 

4438 ).sum() 

4439 ) 

4440 join = StimulusJoin( 

4441 readings=len(readings), 

4442 by_trial=by_trial, 

4443 by_text=matched - by_trial, 

4444 ambiguous_texts=ambiguous, 

4445 ambiguous_readings=ambiguous_readings, 

4446 missing_screen_readings=missing_screen_readings, 

4447 missing_screens=missing_screens, 

4448 text_mismatches=mismatches, 

4449 mismatch_example=mismatch_example, 

4450 multipart=bool(screen), 

4451 ) 

4452 readings[_STIMULUS_KEY] = stimulus 

4453 return join, readings, reading_key 

4454 

4455 

4456def plan_stimulus_join( 

4457 words: pd.DataFrame, fixations: pd.DataFrame 

4458) -> StimulusJoin | None: 

4459 """How :func:`broadcast_stimulus_words` attaches ``words`` (DATA-49). 

4460 

4461 ``None`` when there is nothing to join: the words carry a Participant ID 

4462 (so they are not stimulus-level), or either frame is empty. Otherwise the 

4463 :class:`StimulusJoin` it would make — ``key=None`` when it would refuse. 

4464 ``harmonize_frames`` runs other fixups first (BUG-59's zero padding), so 

4465 for the join a load actually made, use :func:`harmonize_frames_with_join`. 

4466 """ 

4467 if STIMULUS_WORDS_FLAG not in words.columns or words.empty or fixations.empty: 

4468 return None 

4469 return _plan_stimulus_join(words, fixations)[0] 

4470 

4471 

4472def _broadcast_stimulus_words( 

4473 words: pd.DataFrame, fixations: pd.DataFrame 

4474) -> tuple[pd.DataFrame, StimulusJoin | None]: 

4475 """:func:`broadcast_stimulus_words`, plus the join it made (``None`` when 

4476 it made none: per-reader words, or nothing to broadcast against).""" 

4477 if STIMULUS_WORDS_FLAG not in words.columns: 

4478 return words, None 

4479 if words.empty or fixations.empty: 

4480 words = words.drop(columns=[STIMULUS_WORDS_FLAG]) 

4481 # No fixations to broadcast across (e.g. a words-only dataset): there's a 

4482 # single anonymous reader, so give the placeholder a real synthetic id. 

4483 if not words.empty: 

4484 words = words.copy() 

4485 words["participant_id"] = SYNTHETIC_PARTICIPANT 

4486 return words, None 

4487 join, readings, reading_key = _plan_stimulus_join(words, fixations) 

4488 # A multipart dataset needs boxes for every screen it has fixations on 

4489 # (`validate_matching_parts` would reject it a step later as "orphan 

4490 # screens"), so a partial join is refused here, with the reasons. 

4491 if join.key is None or (join.multipart and join.unmatched): 

4492 raise StimulusJoinError(join.problem()) 

4493 if join.needs_warning: 

4494 # The API and the CLI have no page to put this on, and a script would 

4495 # otherwise get readings with no boxes, or boxes whose Text ID 

4496 # disagrees with the reading's, without a word. 

4497 warnings.warn(join.describe(), StimulusJoinWarning, stacklevel=4) 

4498 else: 

4499 _LOGGER.info(join.describe()) 

4500 screen = reading_key[2:] 

4501 on = [_STIMULUS_KEY, *screen] 

4502 # The AOI table supplies the boxes (and its own Text ID); the reading 

4503 # supplies its identity — the AOI table's trial id is the stimulus', not 

4504 # the reading's, and it stays on each copy as `_aoi_trial_id`. 

4505 stimulus = words.drop( 

4506 columns=[c for c in (STIMULUS_WORDS_FLAG, AOI_TRIAL_ID) if c in words] 

4507 + ["participant_id"] 

4508 ).rename(columns={"trial_id": _STIMULUS_KEY}) 

4509 # Only the key must be present: a blank screen pairs with a blank screen. 

4510 stimulus = stimulus.dropna(subset=[_STIMULUS_KEY]) 

4511 stimulus[_STIMULUS_KEY] = stimulus[_STIMULUS_KEY].astype(str) 

4512 pairs = readings[[*reading_key, _STIMULUS_KEY]].dropna(subset=[_STIMULUS_KEY]) 

4513 pairs = pairs.assign( 

4514 participant_id=pairs["participant_id"].astype(str), 

4515 trial_id=pairs["trial_id"].astype(str), 

4516 ) 

4517 out = stimulus.merge(pairs, on=on, how="inner").rename( 

4518 columns={_STIMULUS_KEY: AOI_TRIAL_ID} 

4519 ) 

4520 if "unique_trial_id" in out.columns: 

4521 # The reading's id *is* its unique trial id (BUG-58). 

4522 out["unique_trial_id"] = out["trial_id"] 

4523 return out, join 

4524 

4525 

4526def broadcast_stimulus_words( 

4527 words: pd.DataFrame, fixations: pd.DataFrame 

4528) -> pd.DataFrame: 

4529 """Give every reading its own copy of the stimulus-level boxes of its text. 

4530 

4531 Datasets like PoTeC ship word/AoI tables per *text* (no participant 

4532 column) while fixations are per participant × text. After normalization, 

4533 such words carry the ``_stimulus_words`` flag; this gives each reading — 

4534 each ``(participant_id, trial_id[, screen_id])`` of the fixations — the 

4535 boxes of the AOI table's trial it belongs to, stamped with that reading's 

4536 own ids (the AOI trial it came from kept as ``_aoi_trial_id``), so 

4537 downstream (participant, trial) filtering works unchanged. Words for texts 

4538 nobody read are dropped. 

4539 

4540 DATA-49: one rule, per reading, whatever the trial ids look like. A reading 

4541 takes the boxes of the first AOI trial it finds by its own trial id (ids 

4542 shared across readers), by the id it had before a repeat's ``_r2`` suffix 

4543 (``_base_trial_id``, BUG-57), or by its **Text ID** (trial ids that embed 

4544 the reader). The Text-ID route needs a real fixations Text ID (mapped, not 

4545 the trial-id fallback — ``_text_id_mapped``), and never uses one the AOI 

4546 table gives to several trials. A trial-id match always stands; mapped Text 

4547 IDs that disagree with it are counted and warned about, never redirected. 

4548 When no reading finds any — or, on a multipart dataset, when any 

4549 reading screen finds none — it raises :class:`StimulusJoinError` rather than 

4550 return a dataset without them, and when only some do it warns 

4551 (:class:`StimulusJoinWarning`). One vectorised merge — it runs on 

4552 million-row corpora. 

4553 

4554 No-op for ordinary per-participant word tables. With no fixations to 

4555 broadcast across (a words-only dataset) the rows get the synthetic reader.""" 

4556 return _broadcast_stimulus_words(words, fixations)[0] 

4557 

4558 

4559def repair_stranded_stimulus_words( 

4560 words: pd.DataFrame, fixations: pd.DataFrame 

4561) -> tuple[pd.DataFrame, pd.DataFrame] | None: 

4562 """DATA-39 — re-broadcast a stored AOI table the old ✅ Save changes stranded. 

4563 

4564 Before DATA-39 was fixed, saving an edit to a dataset whose AOI table has no 

4565 participant column left every word on the ``""`` placeholder reader with 

4566 the ``_stimulus_words`` flag still set, so no trial found its boxes. A 

4567 *stored* frame can only carry that flag through that bug — 

4568 ``broadcast_stimulus_words`` always drops it — so its presence is the 

4569 diagnosis, and running the broadcast it missed is the repair. Returns the 

4570 repaired ``(words, fixations)``, or ``None`` when there is nothing to repair 

4571 or the frames will not harmonize (the dataset is then left as it was). 

4572 """ 

4573 if not isinstance(words, pd.DataFrame) or STIMULUS_WORDS_FLAG not in words.columns: 

4574 return None 

4575 has_fixations = isinstance(fixations, pd.DataFrame) and not fixations.empty 

4576 try: 

4577 repaired, harmonized = harmonize_frames( 

4578 words, fixations if has_fixations else empty_fixations_frame() 

4579 ) 

4580 except Exception: # a repair must never break the load 

4581 _LOGGER.warning( 

4582 "Could not repair a stored Words table left on the placeholder " 

4583 "participant; press Save changes on the Edit dataset screen to retry.", 

4584 exc_info=True, 

4585 ) 

4586 return None 

4587 if repaired.empty: 

4588 return None 

4589 return repaired, (harmonized if has_fixations else fixations) 

4590 

4591 

4592def fill_fixation_xy_from_words( 

4593 fixations: pd.DataFrame, words: pd.DataFrame 

4594) -> pd.DataFrame: 

4595 """Fill missing fixation coordinates from the fixated word's box center. 

4596 

4597 AOI-sequence datasets record *which* word/character each fixation landed 

4598 on but not the pixel position. When normalized fixations have NaN x/y and 

4599 a ``word_id``, place them at the center of the matching word box (keyed by 

4600 participant_id + trial_id + word_id). Fixations whose word_id matches no 

4601 box keep NaN coordinates. Rows that already have coordinates are left 

4602 untouched.""" 

4603 if fixations.empty or words.empty: 

4604 return fixations 

4605 missing = fixations["x"].isna() | fixations["y"].isna() 

4606 if not missing.any() or "word_id" not in fixations.columns: 

4607 return fixations 

4608 from .measures import word_box_bounds 

4609 

4610 # The interest area's own centre (BUG-83), which is inside the box the 

4611 # assignment will then test it against. 

4612 x0, y0, x1, y1 = word_box_bounds(words) 

4613 keys = grouping_columns(words, include_word=True) 

4614 centers = words[keys].copy() 

4615 centers["_word_cx"] = (x0 + x1) / 2.0 

4616 centers["_word_cy"] = (y0 + y1) / 2.0 

4617 centers = centers.drop_duplicates(keys) 

4618 merged = fixations[keys].merge(centers, on=keys, how="left") 

4619 fixations = fixations.copy() 

4620 fill = missing.to_numpy() 

4621 fixations.loc[fill, "x"] = merged["_word_cx"].to_numpy()[fill] 

4622 fixations.loc[fill, "y"] = merged["_word_cy"].to_numpy()[fill] 

4623 return fixations 

4624 

4625 

4626def _reconcile_participant_asymmetry( 

4627 words: pd.DataFrame, fixations: pd.DataFrame 

4628) -> pd.DataFrame: 

4629 """Re-key word boxes to the synthetic participant when the fixations have no 

4630 participant but the words do. 

4631 

4632 With participant now optional per table, a fixations table can be 

4633 participant-less (every row stamped ``SYNTHETIC_PARTICIPANT``) while the words 

4634 table still carries real participant ids. The trial picker keys off the 

4635 fixations, so it offers ``(all)`` — but the boxes are keyed by the real ids 

4636 and ``extract_trial`` then finds none, rendering fixations with no text. Stamp 

4637 the words with the synthetic id (dropping the now-duplicate per-reader boxes) 

4638 so they line up. No-op unless the fixations are entirely synthetic and the 

4639 words are not — the stimulus-words broadcast already covers the reverse.""" 

4640 if words.empty or fixations.empty or "participant_id" not in words.columns: 

4641 return words 

4642 if set(fixations["participant_id"].unique()) != {SYNTHETIC_PARTICIPANT}: 

4643 return words 

4644 word_parts = set(words["participant_id"].unique()) 

4645 if not word_parts or word_parts == {SYNTHETIC_PARTICIPANT}: 

4646 return words 

4647 words = words.copy() 

4648 words["participant_id"] = SYNTHETIC_PARTICIPANT 

4649 subset = grouping_columns(words, include_word=True) 

4650 if subset: 

4651 words = words.drop_duplicates(subset=subset) 

4652 return words 

4653 

4654 

4655_WORD_ID_AGGS = ["min", "max", "nunique"] 

4656 

4657 

4658def _key_frame(frame: pd.DataFrame, ids: pd.Series) -> pd.DataFrame: 

4659 """``(participant_id, trial_id, _id)`` view of ``frame``, aligned to ``ids``. 

4660 

4661 ``ids`` is a NaN-dropped numeric word-id series taken from ``frame``, so the 

4662 id columns are re-indexed onto its (subset) index. 

4663 """ 

4664 data = { 

4665 column: frame[column].reindex(ids.index) for column in grouping_columns(frame) 

4666 } 

4667 data["_id"] = ids 

4668 return pd.DataFrame(data) 

4669 

4670 

4671def detect_word_id_offset(words: pd.DataFrame, fixations: pd.DataFrame) -> int: 

4672 """Detect a 1-based fixation ``word_id`` against 0-based word boxes (BUG-8). 

4673 

4674 Some exports (the bundled OneStop demo among them) number the fixation 

4675 report's word column ``1..N`` while the interest-area table numbers its rows 

4676 ``0..N-1``, so every fixation's pre-assigned ``word_id`` points at the *next* 

4677 word. ``measures.assign_fixations_to_words`` keeps existing ids, so the 

4678 computed reading measures then attach to the wrong words. 

4679 

4680 Returns the offset to **subtract** from the fixation word ids: ``1`` when the 

4681 shift is unambiguous, ``0`` (by far the common case) otherwise. A false 

4682 positive silently corrupts a correct dataset, so the test is deliberately 

4683 strict — every condition below must hold across the whole dataset: 

4684 

4685 * both frames carry whole-number ``word_id`` values, 

4686 * the words ids start at ``0`` and the fixation ids start at ``1``, 

4687 * *every* trial present in both frames has 0-based, gap-free word ids and no 

4688 fixation id below ``1`` or more than one past its last word, and 

4689 * at least one trial actually overflows by exactly one 

4690 (``max fixation id == max word id + 1``) — without that there is no 

4691 evidence of a shift, just a reader who never looked at the first word. 

4692 """ 

4693 if words is None or fixations is None or words.empty or fixations.empty: 

4694 return 0 

4695 keys = grouping_columns(words) 

4696 if keys != grouping_columns(fixations): 

4697 return 0 

4698 needed = set(keys) | {"word_id"} 

4699 if not needed.issubset(words.columns) or not needed.issubset(fixations.columns): 

4700 return 0 

4701 w_ids = pd.to_numeric(words["word_id"], errors="coerce").dropna() 

4702 f_ids = pd.to_numeric(fixations["word_id"], errors="coerce").dropna() 

4703 if w_ids.empty or f_ids.empty: 

4704 return 0 

4705 # Fractional ids (character-level indices, say) aren't a word numbering we 

4706 # can reason about. 

4707 if not (w_ids % 1 == 0).all() or not (f_ids % 1 == 0).all(): 

4708 return 0 

4709 if float(w_ids.min()) != 0.0 or float(f_ids.min()) != 1.0: 

4710 return 0 

4711 # Project onto (keys + id) rather than .assign()-ing onto the source frames: 

4712 # the OneStop words/fixations tables are wide, and this runs on every load. 

4713 w_stats = ( 

4714 _key_frame(words, w_ids).groupby(keys, sort=False)["_id"].agg(_WORD_ID_AGGS) 

4715 ) 

4716 f_stats = ( 

4717 _key_frame(fixations, f_ids) 

4718 .groupby(keys, sort=False)["_id"] 

4719 .agg(["min", "max"]) 

4720 ) 

4721 joined = w_stats.join(f_stats, how="inner", lsuffix="_w", rsuffix="_f") 

4722 if joined.empty: 

4723 return 0 

4724 zero_based = joined["min_w"] == 0 

4725 gap_free = joined["nunique"] == joined["max_w"] + 1 

4726 in_range = joined["min_f"] >= 1 

4727 overflow = joined["max_f"] == joined["max_w"] + 1 

4728 runaway = joined["max_f"] > joined["max_w"] + 1 

4729 if not (zero_based.all() and gap_free.all() and in_range.all()): 

4730 return 0 

4731 if runaway.any() or not overflow.any(): 

4732 return 0 

4733 return 1 

4734 

4735 

4736def correct_word_id_offset( 

4737 words: pd.DataFrame, fixations: pd.DataFrame, *, offset: int | None = None 

4738) -> pd.DataFrame: 

4739 """Shift fixation ``word_id`` back onto the words table when it's 1-based. 

4740 

4741 No-op unless :func:`detect_word_id_offset` finds an unambiguous shift. 

4742 Renumbering someone's ids is never silent — it's logged at INFO, which 

4743 `debug_log.install_log_capture` surfaces in the in-app 🐛 Debug panel. Not a 

4744 `st.warning`, and not a WARNING either (BUG-76): the bundled demo corpus 

4745 trips this on *every* load, so a banner would be permanent furniture on the 

4746 landing view, and a WARNING was the first line `render --sample`, 

4747 `load_sample_data()` and the README quickstart printed to a new user's 

4748 terminal — about a correction that needs nothing from them. 

4749 

4750 ``offset`` is :func:`detect_word_id_offset`'s answer when the caller 

4751 already asked it (DATA-66's harmonize report). 

4752 """ 

4753 if offset is None: 

4754 offset = detect_word_id_offset(words, fixations) 

4755 if not offset: 

4756 return fixations 

4757 fixations = fixations.copy() 

4758 fixations["word_id"] = pd.to_numeric(fixations["word_id"], errors="coerce") - offset 

4759 _LOGGER.info( 

4760 "The fixation report's word ids are numbered from 1 while the word " 

4761 "boxes are numbered from 0, so every fixation pointed at the next word. " 

4762 "Shifted the fixation word ids down by %d to line the two tables up.", 

4763 offset, 

4764 ) 

4765 return fixations 

4766 

4767 

4768def _pad_ids(frame: pd.DataFrame, column: str, mapping: dict) -> pd.DataFrame: 

4769 """``frame`` with ``column`` respelled by a zero-padding ``mapping``. 

4770 

4771 Padding a trial id also pads what was copied from it: a repeat's 

4772 ``_base_trial_id``, and a fallback ``text_id`` (no Text ID mapped, so it 

4773 *is* the trial id) — left at "7" beside a trial "007" it would read as a 

4774 real Text ID of its own (DATA-49).""" 

4775 frame = frame.copy() 

4776 if column == "trial_id": 

4777 if "text_id" in frame.columns: 

4778 copied = _as_key(frame["text_id"]) == _as_key(frame["trial_id"]) 

4779 if TEXT_ID_MAPPED in frame.columns: 

4780 copied &= ~frame[TEXT_ID_MAPPED].fillna(False).astype(bool) 

4781 frame.loc[copied, "text_id"] = frame.loc[copied, "text_id"].replace(mapping) 

4782 if BASE_TRIAL_ID in frame.columns: 

4783 # A repeat is joined by the id it was suffixed from. 

4784 frame[BASE_TRIAL_ID] = frame[BASE_TRIAL_ID].replace(mapping) 

4785 frame[column] = frame[column].replace(mapping) 

4786 return frame 

4787 

4788 

4789def _restore_zero_padding( 

4790 words: pd.DataFrame, 

4791 fixations: pd.DataFrame, 

4792 padded: list[tuple[str, str]] | None = None, 

4793) -> tuple[pd.DataFrame, pd.DataFrame]: 

4794 """Spell a zero-padded id the same way in both frames (BUG-59). 

4795 

4796 A CSV read ``007`` as 7 while a Parquet table kept "007", and the two 

4797 tables then shared no participant — every fixation drew over no text. When 

4798 padding is the only difference (:func:`zero_padding_map`), the side that 

4799 lost its zeros is given them back, and the rename is logged — and recorded 

4800 in ``padded`` as ``(table, column)`` when given. 

4801 """ 

4802 if words.empty or fixations.empty: 

4803 return words, fixations 

4804 columns = ["trial_id", "text_id"] 

4805 if STIMULUS_WORDS_FLAG not in words.columns: 

4806 columns.insert(0, "participant_id") 

4807 # The second respelling: one table stored before composite ids escaped a 

4808 # `_` inside a part (`compose_id`), the other composed since — a table 

4809 # added on ✏️ Edit dataset to a dataset restored from the recovery cache. 

4810 respellings = ( 

4811 (zero_padding_map, "without the zero-padding the other table uses"), 

4812 (composite_respelling_map, "with the other spelling of a composite id"), 

4813 ) 

4814 for column in columns: 

4815 if column not in words.columns or column not in fixations.columns: 

4816 continue 

4817 w_ids, f_ids = words[column].unique(), fixations[column].unique() 

4818 for respell, how in respellings: 

4819 for frame_name, ids, reference in ( 

4820 ("words", w_ids, f_ids), 

4821 ("fixations", f_ids, w_ids), 

4822 ): 

4823 mapping = respell(ids, reference) 

4824 if not mapping: 

4825 continue 

4826 if frame_name == "words": 

4827 words = _pad_ids(words, column, mapping) 

4828 else: 

4829 fixations = _pad_ids(fixations, column, mapping) 

4830 if padded is not None: 

4831 padded.append((frame_name, column)) 

4832 _LOGGER.info( 

4833 "The %s table spelled %d %s value(s) %s (e.g. %r for %r); " 

4834 "matched them up.", 

4835 frame_name, 

4836 len(mapping), 

4837 column, 

4838 how, 

4839 *next(iter(mapping.items()))[::-1], 

4840 ) 

4841 w_ids, f_ids = words[column].unique(), fixations[column].unique() 

4842 break 

4843 return words, fixations 

4844 

4845 

4846def harmonize_frames_with_join( 

4847 words: pd.DataFrame, fixations: pd.DataFrame 

4848) -> tuple[pd.DataFrame, pd.DataFrame, StimulusJoin | None]: 

4849 """Cross-frame fixups applied right after normalization. 

4850 

4851 Match zero-padded ids the two tables spell differently (BUG-59), broadcast 

4852 stimulus-level words across participants, reconcile a participant-less 

4853 fixations table with participant-bearing words, correct a 1-based fixation 

4854 ``word_id`` (BUG-8), then fill missing fixation coordinates from word-box 

4855 centers. Call whenever both frames are available (the API and the app both 

4856 route through this). Also returns the :class:`StimulusJoin` a 

4857 stimulus-level AOI table was attached by — ``None`` for a per-reader one — 

4858 which the add-dataset wizard states (DATA-49).""" 

4859 words, fixations, join, _rewrites = harmonize_frames_reporting(words, fixations) 

4860 return words, fixations, join 

4861 

4862 

4863#: DATA-66: a column whose values normalization or the fixups changed, as 

4864#: ``(table, column, how)`` — ``how`` follows the source column's name in its 

4865#: label ("CURRENT_FIX_INTEREST_AREA_ID − 1"), since a header the file used must 

4866#: not sit over values the file never held. ``column_names.with_rewrites`` 

4867#: applies them. 

4868Rewrite = tuple[str, str, str] 

4869 

4870 

4871def harmonize_frames_reporting( 

4872 words: pd.DataFrame, fixations: pd.DataFrame 

4873) -> tuple[pd.DataFrame, pd.DataFrame, StimulusJoin | None, tuple[Rewrite, ...]]: 

4874 """:func:`harmonize_frames_with_join`, also saying which columns' values it 

4875 changed (:data:`Rewrite`): ids it zero-padded (BUG-59), a fixation 

4876 ``word_id`` shifted onto 0-based boxes (BUG-8), fixation positions filled 

4877 from word boxes, and a repeated reading's ``_rN`` trial id (BUG-57).""" 

4878 from .preprocessing import add_text_direction 

4879 

4880 rewrites: list[Rewrite] = [] 

4881 padded: list[tuple[str, str]] = [] 

4882 words, fixations = _restore_zero_padding(words, fixations, padded) 

4883 rewrites += [ 

4884 (table, column, ", respelled to match the other table") 

4885 for table, column in padded 

4886 ] 

4887 words, join = _broadcast_stimulus_words(words, fixations) 

4888 words = add_text_direction(words) 

4889 words = _reconcile_participant_asymmetry(words, fixations) 

4890 words = normalize_screen_identity(words) 

4891 fixations = normalize_screen_identity(fixations) 

4892 validate_matching_parts(words, fixations) 

4893 offset = detect_word_id_offset(words, fixations) 

4894 fixations = correct_word_id_offset(words, fixations, offset=offset) 

4895 if offset: 

4896 rewrites.append(("fixations", "word_id", f" − {offset}")) 

4897 blank = ( 

4898 fixations[["x", "y"]].isna().sum() 

4899 if {"x", "y"} <= set(fixations.columns) 

4900 else None 

4901 ) 

4902 fixations = fill_fixation_xy_from_words(fixations, words) 

4903 if blank is not None: 

4904 filled = blank - fixations[["x", "y"]].isna().sum() 

4905 rewrites += [ 

4906 ("fixations", axis, ", blanks filled from the word boxes") 

4907 for axis in ("x", "y") 

4908 if filled[axis] > 0 

4909 ] 

4910 if ( 

4911 BASE_TRIAL_ID in fixations.columns 

4912 and (_as_key(fixations["trial_id"]) != _as_key(fixations[BASE_TRIAL_ID])) 

4913 .where(fixations[BASE_TRIAL_ID].notna(), False) 

4914 .any() 

4915 ): 

4916 rewrites.append(("fixations", "trial_id", " + _rN for a repeated trial")) 

4917 return words, fixations, join, tuple(rewrites) 

4918 

4919 

4920def harmonize_frames( 

4921 words: pd.DataFrame, fixations: pd.DataFrame 

4922) -> tuple[pd.DataFrame, pd.DataFrame]: 

4923 """:func:`harmonize_frames_with_join` without the join report.""" 

4924 words, fixations, _join = harmonize_frames_with_join(words, fixations) 

4925 return words, fixations 

4926 

4927 

4928def _disambiguate_repeated_readings( 

4929 df: pd.DataFrame, 

4930 source: pd.DataFrame, 

4931 trial_col: str, 

4932 *, 

4933 record_base: bool = False, 

4934) -> pd.DataFrame: 

4935 """Suffix `trial_id` with `_r2`, `_r3` … when a participant read the same 

4936 paragraph more than once. 

4937 

4938 OneStop L2's per-pid parquet shards don't carry a `unique_trial_id` column, 

4939 so the schema-inference fallback uses `unique_paragraph_id` — but that's 

4940 the same string for both readings of a repeated-reading trial. Without 

4941 this fix, the two readings' fixations collapse into one scanpath (and into 

4942 one row of the trial picker), which is what the cached PNG thumbnails 

4943 (which filter on TRIAL_INDEX) correctly avoid. We rank by TRIAL_INDEX so 

4944 the chronologically-first reading keeps its original id; later readings 

4945 get `_r2`, `_r3`, … appended. 

4946 

4947 Groups on the already-computed ``df["participant_id"]`` (1:1 with ``source``), 

4948 so a composite participant id is handled without recomputing the join. 

4949 """ 

4950 if trial_col == "unique_trial_id": 

4951 return df 

4952 idx_col = next( 

4953 (c for c in ("TRIAL_INDEX", "trial_index") if c in source.columns), None 

4954 ) 

4955 if idx_col is None: 

4956 return df 

4957 grouper = pd.DataFrame( 

4958 { 

4959 "_pk": df["participant_id"].to_numpy(), 

4960 "_tc": source[trial_col].astype(str).to_numpy(), 

4961 "_idx": source[idx_col].to_numpy(), 

4962 } 

4963 ) 

4964 # A reading with no index of its own keeps its id unsuffixed rather than 

4965 # crashing the cast (BUG-56). 

4966 rank = ( 

4967 grouper.groupby(["_pk", "_tc"])["_idx"] 

4968 .rank(method="dense") 

4969 .fillna(1) 

4970 .astype(int) 

4971 .to_numpy() 

4972 ) 

4973 base = df["trial_id"] 

4974 df["trial_id"] = [ 

4975 tid if r == 1 else f"{tid}_r{r}" for tid, r in zip(base.to_numpy(), rank) 

4976 ] 

4977 if record_base and (rank > 1).any(): 

4978 # What the reading was recorded under, so a table keyed by the stimulus 

4979 # still finds a repeat's boxes (BUG-57 / DATA-49's trial join). 

4980 df[BASE_TRIAL_ID] = base 

4981 return df 

4982 

4983 

4984def has_explicit_trial_index(frame: pd.DataFrame) -> bool: 

4985 """True when the data already carries a per-trial index column.""" 

4986 return any(c in frame.columns for c in ("trial_index", "TRIAL_INDEX")) 

4987 

4988 

4989def trial_order_label(frame: pd.DataFrame) -> str: 

4990 """The axis title for :func:`derive_trial_index`'s values: the column it 

4991 reads, or how it ordered the trials without one (#374 F35 — the demo 

4992 carries both ``trial_index`` and ``TRIAL_INDEX``, which differ).""" 

4993 for col in ("trial_index", "TRIAL_INDEX"): 

4994 if col in frame.columns: 

4995 return f"Trial order ({col})" 

4996 if "timestamp_ms" in frame.columns: 

4997 return "Trial order (by fixation time)" 

4998 return "Trial order (as listed)" 

4999 

5000 

5001def derive_trial_index(frame: pd.DataFrame) -> pd.Series: 

5002 """Per-participant 1-based trial order, aligned to ``frame``'s rows. 

5003 

5004 Prefers an existing ``trial_index`` / ``TRIAL_INDEX`` column (the order the 

5005 data already records). Otherwise it ranks each participant's trials by their 

5006 earliest ``timestamp_ms`` (falling back to first-appearance order when no 

5007 timestamps) and numbers them 1, 2, 3, …. Used by the Corpus Analysis tab to 

5008 plot a metric as a function of where the trial fell in the session. Returns a 

5009 float Series (NaN where the index can't be determined).""" 

5010 if frame.empty or not {"participant_id", "trial_id"} <= set(frame.columns): 

5011 return pd.Series([np.nan] * len(frame), index=frame.index, dtype="float64") 

5012 for col in ("trial_index", "TRIAL_INDEX"): 

5013 if col in frame.columns: 

5014 return pd.to_numeric(frame[col], errors="coerce") 

5015 if "timestamp_ms" in frame.columns: 

5016 order_key = frame.groupby(["participant_id", "trial_id"])[ 

5017 "timestamp_ms" 

5018 ].transform("min") 

5019 else: 

5020 # First-appearance order: row position of each trial's first row. 

5021 order_key = pd.Series(range(len(frame)), index=frame.index) 

5022 order_key = ( 

5023 frame.assign(_k=order_key) 

5024 .groupby(["participant_id", "trial_id"])["_k"] 

5025 .transform("min") 

5026 ) 

5027 per_trial = ( 

5028 frame[["participant_id", "trial_id"]] 

5029 .assign(_k=pd.to_numeric(order_key, errors="coerce")) 

5030 .drop_duplicates(["participant_id", "trial_id"]) 

5031 .sort_values(["participant_id", "_k"]) 

5032 ) 

5033 per_trial["_idx"] = per_trial.groupby("participant_id").cumcount() + 1 

5034 merged = frame[["participant_id", "trial_id"]].merge( 

5035 per_trial[["participant_id", "trial_id", "_idx"]], 

5036 on=["participant_id", "trial_id"], 

5037 how="left", 

5038 ) 

5039 return pd.Series( 

5040 pd.to_numeric(merged["_idx"], errors="coerce").to_numpy(), 

5041 index=frame.index, 

5042 dtype="float64", 

5043 ) 

5044 

5045 

5046# --------------------------------------------------------------------------- 

5047# Optional-field registry. Drives (a) which known optional source columns are 

5048# carried into the normalized frame and (b) the setup wizard's opt-out checklist. 

5049# Each entry: (source, dest, kind, category) where `kind` ∈ 

5050# {numeric, string, boolean, passthrough} and `category` ∈ 

5051# {measure, linguistic, meta} groups the fields in the UI. Matched by exact 

5052# source name (same as the legacy keep-lists this replaced). 

5053# --------------------------------------------------------------------------- 

5054WORD_OPTIONAL_FIELDS = [ 

5055 ("IA_FIRST_FIXATION_DURATION", "first_fixation_ms", "numeric", "measure"), 

5056 ("IA_DWELL_TIME", "total_fixation_duration_ms", "numeric", "measure"), 

5057 ("IA_FIRST_RUN_DWELL_TIME", "first_pass_gaze_duration_ms", "numeric", "measure"), 

5058 ( 

5059 "IA_SECOND_RUN_DWELL_TIME", 

5060 "second_pass_duration_ms", 

5061 "numeric", 

5062 "measure", 

5063 ), 

5064 # Compatibility aliases retained for existing datasets/API consumers; the 

5065 # PRE-4 canonical field above is the one measure computation consults. 

5066 ( 

5067 "IA_SECOND_RUN_DWELL_TIME", 

5068 "higher_pass_fixation_duration_ms", 

5069 "numeric", 

5070 "measure", 

5071 ), 

5072 ("IA_LAST_RUN_DWELL_TIME", "last_run_dwell_time_ms", "numeric", "measure"), 

5073 ("IA_FIXATION_COUNT", "n_fixations", "numeric", "measure"), 

5074 ("IA_SKIP", "skip_flag", "boolean", "measure"), 

5075 ( 

5076 "IA_REGRESSION_IN_COUNT", 

5077 "number_of_regressions_in", 

5078 "numeric", 

5079 "measure", 

5080 ), 

5081 ("IA_REGRESSION_IN_COUNT", "regression_in_count", "numeric", "measure"), 

5082 ("IA_REGRESSION_OUT_COUNT", "regression_out_count", "numeric", "measure"), 

5083 ("IA_REGRESSION_IN", "regression_in_flag", "boolean", "measure"), 

5084 ("IA_REGRESSION_OUT", "regression_out_flag", "boolean", "measure"), 

5085 ( 

5086 "IA_REGRESSION_PATH_DURATION", 

5087 "regression_path_duration_ms", 

5088 "numeric", 

5089 "measure", 

5090 ), 

5091 ("TRIAL_DWELL_TIME", "trial_dwell_time_ms", "numeric", "measure"), 

5092 ("TRIAL_FIXATION_COUNT", "trial_fixation_count", "numeric", "measure"), 

5093 ("TRIAL_IA_COUNT", "trial_ia_count", "numeric", "measure"), 

5094 # A property of the word, not of the reading — and Corpus Analysis' "Word 

5095 # length" feature, so it stays pre-kept with the other linguistic fields. 

5096 ("word_length", "word_length", "numeric", "linguistic"), 

5097 ( 

5098 "word_length_no_punctuation", 

5099 "word_length_no_punctuation", 

5100 "numeric", 

5101 "linguistic", 

5102 ), 

5103 ("gpt2_surprisal", "gpt2_surprisal", "numeric", "linguistic"), 

5104 ("wordfreq_frequency", "wordfreq_frequency", "numeric", "linguistic"), 

5105 ("subtlex_frequency", "subtlex_frequency", "numeric", "linguistic"), 

5106 ("universal_pos", "universal_pos", "string", "linguistic"), 

5107 ("ptb_pos", "ptb_pos", "string", "linguistic"), 

5108 ("Reduced_POS", "reduced_pos", "string", "linguistic"), 

5109 ("dependency_relation", "dependency_relation", "string", "linguistic"), 

5110 ("morphological_features", "morphological_features", "string", "linguistic"), 

5111 ("entity_type", "entity_type", "string", "linguistic"), 

5112 ("head_word_index", "head_word_index", "numeric", "linguistic"), 

5113 ("distance_to_head", "distance_to_head", "numeric", "linguistic"), 

5114 ("left_dependents_count", "left_dependents_count", "numeric", "linguistic"), 

5115 ("right_dependents_count", "right_dependents_count", "numeric", "linguistic"), 

5116 ("sentence_id", "sentence_id", "passthrough", "linguistic"), 

5117 ("SENTENCE_ID", "sentence_id", "passthrough", "linguistic"), 

5118 ("right_to_left", "right_to_left", "boolean", "meta"), 

5119 ("RIGHT_TO_LEFT", "right_to_left", "boolean", "meta"), 

5120 (SOURCE_FILE_COLUMN, SOURCE_FILE_COLUMN, "passthrough", "meta"), 

5121 ("TRIAL_INDEX", "TRIAL_INDEX", "passthrough", "meta"), 

5122 ("trial_index", "trial_index", "passthrough", "meta"), 

5123 ("article_batch", "article_batch", "passthrough", "meta"), 

5124 ("article_id", "article_id", "passthrough", "meta"), 

5125 ("difficulty_level", "difficulty_level", "passthrough", "meta"), 

5126 ("article_title", "article_title", "passthrough", "meta"), 

5127 ("question", "question", "passthrough", "meta"), 

5128 ("question_preview", "question_preview", "boolean", "meta"), 

5129 ("selected_answer", "selected_answer", "passthrough", "meta"), 

5130 ("is_correct", "is_correct", "passthrough", "meta"), 

5131 ("repeated_reading_trial", "repeated_reading_trial", "boolean", "meta"), 

5132 ("critical_span_indices", "critical_span_indices", "passthrough", "meta"), 

5133 ("distractor_span_indices", "distractor_span_indices", "passthrough", "meta"), 

5134 ("aspan_ind_start", "aspan_ind_start", "passthrough", "meta"), 

5135 ("aspan_ind_end", "aspan_ind_end", "passthrough", "meta"), 

5136 ("dspan_ind_start", "dspan_ind_start", "passthrough", "meta"), 

5137 ("dspan_ind_end", "dspan_ind_end", "passthrough", "meta"), 

5138 ("is_in_aspan", "is_in_aspan", "boolean", "meta"), 

5139 ("is_in_dspan", "is_in_dspan", "boolean", "meta"), 

5140 # MultiplEYE side-data (also see FIX_OPTIONAL_FIELDS): the comprehension 

5141 # questions JSON + the per-trial stimulus-image path + the genre facet, kept 

5142 # so the panels / image layer can read them off the word frame too. 

5143 ("comprehension_questions", "comprehension_questions", "passthrough", "meta"), 

5144 ("image_path", "image_path", "passthrough", "meta"), 

5145 ("image_x", "image_x", "numeric", "meta"), 

5146 ("image_y", "image_y", "numeric", "meta"), 

5147 # Stimulus typeface (size in monitor px + CSS family) the images were rendered 

5148 # with — the app snaps its font controls to these so the reading text matches. 

5149 ("stimulus_font_px", "stimulus_font_px", "numeric", "meta"), 

5150 ("stimulus_font_family", "stimulus_font_family", "passthrough", "meta"), 

5151 ("genre", "genre", "string", "meta"), 

5152 # DATA-24 multipart screens: what a screen *is* (`reading` / `question` on 

5153 # MultiplEYE — a trial mixes both, and their fixations have different 

5154 # provenance), and which block of a question screen a word belongs to 

5155 # (`stem` / `target` / `distractor_a`… — a per-word facet). 

5156 ("screen_kind", "screen_kind", "passthrough", "meta"), 

5157 ("aoi_block", "aoi_block", "passthrough", "meta"), 

5158 # DATA-27: which tier the EyeGenBench word boxes came from -- 

5159 # "real" | "reconstructed" | "synthesized". Carried so the UI can badge a 

5160 # reconstructed layout rather than pass it off as the original screen. 

5161 ("geometry_source", "geometry_source", "passthrough", "meta"), 

5162] 

5163 

5164FIX_OPTIONAL_FIELDS = [ 

5165 (SOURCE_FILE_COLUMN, SOURCE_FILE_COLUMN, "passthrough", "meta"), 

5166 ("TRIAL_INDEX", "TRIAL_INDEX", "passthrough", "meta"), 

5167 ("trial_index", "trial_index", "passthrough", "meta"), 

5168 ("article_batch", "article_batch", "passthrough", "meta"), 

5169 ("article_id", "article_id", "passthrough", "meta"), 

5170 ("difficulty_level", "difficulty_level", "passthrough", "meta"), 

5171 ("article_title", "article_title", "passthrough", "meta"), 

5172 ("question", "question", "passthrough", "meta"), 

5173 ("selected_answer", "selected_answer", "passthrough", "meta"), 

5174 ("is_correct", "is_correct", "passthrough", "meta"), 

5175 ("repeated_reading_trial", "repeated_reading_trial", "boolean", "meta"), 

5176 ("question_preview", "question_preview", "boolean", "meta"), 

5177 # Per-fixation extras — auto-detected + kept (renamed to canonical so the 

5178 # colour-by / per-fixation filters still find them), but not schema mapping 

5179 # fields. saccade_amplitude is also recomputed from X/Y by measures when 

5180 # absent, so it's never lost. 

5181 ("pass_index", "pass_index", "numeric", "fixation"), 

5182 ("reread", "pass_index", "numeric", "fixation"), 

5183 ("saccade_type", "saccade_type", "string", "fixation"), 

5184 ("NEXT_SAC_DIRECTION", "saccade_type", "string", "fixation"), 

5185 # BUG-25: `saccade_amplitude` is *pixels* — either the source column of that 

5186 # name, or (when absent) the euclidean distance measures.py computes from 

5187 # X/Y. EyeLink's two amplitude columns are **degrees of visual angle**, and 

5188 # they are two *different* saccades — the one leaving this fixation and the 

5189 # one that arrived at it — so each keeps its own canonical name with the unit 

5190 # in it. They are deliberately NOT aliased onto `saccade_amplitude`: that 

5191 # made one column mean px or deg depending on which columns the export 

5192 # happened to carry (~78x apart on the bundled demo), under a hard-coded 

5193 # "px" label. Converting instead would need `pixels_per_degree`, which is 

5194 # ASSUMED on every built-in corpus. 

5195 ("saccade_amplitude", "saccade_amplitude", "numeric", "fixation"), 

5196 ("NEXT_SAC_AMPLITUDE", "next_saccade_amplitude_deg", "numeric", "fixation"), 

5197 ("PREVIOUS_SAC_AMPLITUDE", "prev_saccade_amplitude_deg", "numeric", "fixation"), 

5198 ("eye", "eye", "string", "fixation"), 

5199 ("EYE_USED", "eye", "string", "fixation"), 

5200 ("EYE_TRACKED", "eye", "string", "fixation"), 

5201 ("is_blink", "is_blink", "boolean", "fixation"), 

5202 ("blink", "is_blink", "boolean", "fixation"), 

5203 ("blink_flag", "is_blink", "boolean", "fixation"), 

5204 ("BLINK", "is_blink", "boolean", "fixation"), 

5205 # MultiplEYE trial-level facets + side-data → Trial Info chips / filter 

5206 # facets / the comprehension panel / the stimulus-image layer. All are 

5207 # MultiplEYE-specific source names (carried only when the loader emits them), 

5208 # so they're inert for other corpora. 

5209 ("genre", "genre", "string", "meta"), 

5210 ("session", "session", "string", "meta"), 

5211 ("participant", "participant", "string", "meta"), 

5212 ("is_practice", "is_practice", "boolean", "meta"), 

5213 ("trial_num", "trial_num", "numeric", "meta"), 

5214 ("comprehension_questions", "comprehension_questions", "passthrough", "meta"), 

5215 ("image_path", "image_path", "passthrough", "meta"), 

5216 ("image_x", "image_x", "numeric", "meta"), 

5217 ("image_y", "image_y", "numeric", "meta"), 

5218 ("stimulus_font_px", "stimulus_font_px", "numeric", "meta"), 

5219 ("stimulus_font_family", "stimulus_font_family", "passthrough", "meta"), 

5220 # Reader metadata merged from participant_data.csv (namespaced pp_*). 

5221 ("pp_age", "pp_age", "numeric", "meta"), 

5222 ("pp_gender", "pp_gender", "string", "meta"), 

5223 ("pp_native_language", "pp_native_language", "string", "meta"), 

5224 ("pp_years_education", "pp_years_education", "numeric", "meta"), 

5225 ("pp_education_level", "pp_education_level", "string", "meta"), 

5226 # DATA-24: which kind of screen a fixation happened on (see the word table). 

5227 ("screen_kind", "screen_kind", "passthrough", "meta"), 

5228 # DATA-27: which tier the EyeGenBench word boxes came from -- 

5229 # "real" | "reconstructed" | "synthesized". Carried so the UI can badge a 

5230 # reconstructed layout rather than pass it off as the original screen. 

5231 ("geometry_source", "geometry_source", "passthrough", "meta"), 

5232 # DATA-31: whether EyeGenBench retained the recorded vertical coordinate or 

5233 # placed this fixation at its word box's centre. 

5234 ("fixation_y_source", "fixation_y_source", "passthrough", "meta"), 

5235 # DATA-27: EyeGenBench's own composite trial id, kept for traceability back to the 

5236 # benchmark. Deliberately NOT named `unique_trial_id` — normalize_fixations used to 

5237 # key trial_id on any column with that literal name (BUG-58), and the normalized 

5238 # frame's own `unique_trial_id` is the mapped trial id, so these would not survive. 

5239 ("eyegenbench_trial_id", "eyegenbench_trial_id", "passthrough", "meta"), 

5240] 

5241 

5242 

5243def _schema_source_columns(schema: dict) -> set: 

5244 """Set of raw source column names a normalization schema references.""" 

5245 cols: set = set() 

5246 for value in schema.values(): 

5247 if not value: 

5248 continue 

5249 if isinstance(value, list): 

5250 cols.update(value) 

5251 else: 

5252 cols.add(value) 

5253 return cols 

5254 

5255 

5256def dropped_columns( 

5257 raw: pd.DataFrame, 

5258 *, 

5259 keep: set | None = None, 

5260 schema: dict | None = None, 

5261) -> list: 

5262 """Original source columns discarded during normalization (sorted). 

5263 

5264 Pass ``keep`` (the set handed to ``normalize_words``/``normalize_fixations``, 

5265 i.e. a ``compute_keep_columns`` result) for the union-keep tables, or 

5266 ``schema`` for raw gaze (``normalize_raw_gaze`` keeps the schema-referenced 

5267 columns plus any ``unique_trial_id`` it consults directly). With neither, 

5268 returns ``[]``.""" 

5269 if keep is None and schema is not None: 

5270 keep = _schema_source_columns(schema) | {"unique_trial_id"} 

5271 if keep is None: 

5272 return [] 

5273 return sorted(c for c in raw.columns if c not in keep) 

5274 

5275 

5276# EyeLink writes a missing value as the string ``'.'`` and booleans as ``'0'`` / 

5277# ``'1'``, so a whole flag column arrives as *strings* (BUG-7). A plain 

5278# ``astype(bool)`` then reads every non-empty string as True — including ``'0'`` 

5279# and ``'.'`` — and `regression_in_flag` came out True for every row in the 

5280# bundled demo. Anything a reader would write for "false" or "missing" has to be 

5281# recognised before the cast. 

5282_FALSEY_FLAG_STRINGS = {"", ".", "0", "0.0", "false", "f", "no", "n", "na", "nan", "-"} 

5283 

5284 

5285def coerce_flag(col: pd.Series) -> pd.Series: 

5286 """Coerce a flag-like column to real booleans (BUG-7). 

5287 

5288 Numbers go by ``!= 0``; strings are matched against the sentinels above 

5289 (case-insensitively) rather than by truthiness. NaN / missing is ``False``, 

5290 which is what every downstream flag consumer already assumed. 

5291 """ 

5292 if pd.api.types.is_bool_dtype(col): 

5293 return col.fillna(False).astype(bool) 

5294 numeric = pd.to_numeric(col, errors="coerce") 

5295 if numeric.notna().any(): 

5296 # A numeric-looking column ('0'/'1', 0/1, 0.0/1.0): non-zero is True. 

5297 # Values that didn't parse fall through to the string test below, so a 

5298 # mixed '0'/'1'/'.' column doesn't lose its '.' rows to True. 

5299 parsed = numeric.notna() 

5300 # Build the boolean array positionally: assigning a bool ndarray into a 

5301 # bool Series by mask is deprecated in pandas as a dtype-incompatible set. 

5302 values = (numeric.to_numpy() != 0) & parsed.to_numpy() 

5303 unparsed = (~parsed & col.notna()).to_numpy() 

5304 if unparsed.any(): 

5305 values[unparsed] = ~( 

5306 col[unparsed] 

5307 .astype(str) 

5308 .str.strip() 

5309 .str.lower() 

5310 .isin(_FALSEY_FLAG_STRINGS) 

5311 ).to_numpy() 

5312 result = pd.Series(values, index=col.index, dtype=bool) 

5313 return result 

5314 return ( 

5315 ~col.fillna("").astype(str).str.strip().str.lower().isin(_FALSEY_FLAG_STRINGS) 

5316 ).astype(bool) 

5317 

5318 

5319#: What a supplied reading-measure flag writes for "not recorded" — EyeLink's 

5320#: `.` for a word with no first pass, an empty cell, a spelled-out NA. 

5321_MISSING_FLAG_STRINGS = {"", ".", "na", "nan", "n/a", "-", "none", "null", "<na>"} 

5322 

5323 

5324_TRUE_FLAG_SPELLINGS = {"true", "t", "yes", "y", "1", "1.0"} 

5325_FALSE_FLAG_SPELLINGS = {"false", "f", "no", "n", "0", "0.0"} 

5326_FLAG_SPELLINGS = { 

5327 **dict.fromkeys(_TRUE_FLAG_SPELLINGS, True), 

5328 **dict.fromkeys(_FALSE_FLAG_SPELLINGS, False), 

5329} 

5330 

5331 

5332def coerce_bool_or_na(col: pd.Series) -> pd.Series: 

5333 """A user-supplied true/false column as a nullable boolean. 

5334 

5335 Real booleans stay as they are, numbers go by ``!= 0``, and the strings 

5336 ``true/false``, ``t/f``, ``yes/no``, ``y/n``, ``1/0`` (any case, trimmed) 

5337 are read by their meaning — never by truthiness, under which the string 

5338 ``"False"`` is true. Missing cells and any other spelling are ``<NA>``, so 

5339 a caller decides what an unknown means (round 11). 

5340 """ 

5341 if pd.api.types.is_bool_dtype(col): 

5342 return col.astype("boolean") 

5343 if pd.api.types.is_numeric_dtype(col): 

5344 return (col != 0).astype("boolean").mask(col.isna()) 

5345 out = pd.Series(pd.NA, index=col.index, dtype="boolean") 

5346 is_str = col.map(type).eq(str).to_numpy() 

5347 is_bool = col.map(type).eq(bool).to_numpy() 

5348 if is_bool.any(): 

5349 out[is_bool] = col[is_bool].astype(bool).to_numpy() 

5350 numeric = pd.to_numeric(col.where(~is_str & ~is_bool), errors="coerce") 

5351 is_number = numeric.notna().to_numpy() 

5352 out[is_number] = (numeric[is_number] != 0).to_numpy() 

5353 if is_str.any(): 

5354 spelled = col[is_str].str.strip().str.lower() 

5355 out[is_str] = spelled.map(_FLAG_SPELLINGS).astype("boolean").to_numpy() 

5356 return out 

5357 

5358 

5359def coerce_measure_flag(col: pd.Series) -> pd.Series: 

5360 """A supplied reading-measure flag (skip, regression in/out) as a nullable 

5361 boolean: true, false, or missing. 

5362 

5363 :func:`coerce_flag` reads a missing cell as ``False``, which is right for an 

5364 operational flag (blink, excluded) and wrong for a measure: EyeLink writes 

5365 ``.`` in ``IA_REGRESSION_IN`` for a word with no first pass, where the 

5366 measure is undefined, not "no regression". As ``False`` those rows lowered 

5367 every rate and counted as readers behind it.""" 

5368 flags = coerce_flag(col).astype("boolean") 

5369 if pd.api.types.is_bool_dtype(col) and not col.isna().any(): 

5370 return flags 

5371 missing = col.isna() | col.astype(str).str.strip().str.lower().isin( 

5372 _MISSING_FLAG_STRINGS 

5373 ) 

5374 flags[missing.to_numpy()] = pd.NA 

5375 return flags 

5376 

5377 

5378def _apply_optional_fields( 

5379 df: pd.DataFrame, source: pd.DataFrame, registry: list, keep: set | None 

5380) -> set: 

5381 """Carry registry-listed optional source columns into ``df`` (renamed + 

5382 dtype-coerced). ``keep`` is ``None`` (carry every detected field — the 

5383 backward-compatible default) or a set of *source* column names to limit to. 

5384 Returns the set of source columns actually emitted.""" 

5385 emitted: set = set() 

5386 for src, dest, kind, category in registry: 

5387 if src not in source.columns: 

5388 continue 

5389 if keep is not None and src not in keep: 

5390 continue 

5391 emitted.add(src) 

5392 col = source[src] 

5393 if kind == "numeric": 

5394 df[dest] = _to_number(col) 

5395 elif kind == "string": 

5396 df[dest] = col.astype(str) 

5397 elif kind == "boolean" and category == "measure": 

5398 # A reading measure keeps "not recorded" apart from "false". 

5399 df[dest] = coerce_measure_flag(col) 

5400 elif kind == "boolean": 

5401 df[dest] = coerce_flag(col) 

5402 else: 

5403 df[dest] = col 

5404 return emitted 

5405 

5406 

5407def _carry_extra_columns( 

5408 df: pd.DataFrame, source: pd.DataFrame, keep: set | None, skip: set 

5409) -> None: 

5410 """Carry user-chosen extra ``keep`` source columns through verbatim, skipping 

5411 those already emitted (canonical / registry) or in ``skip``.""" 

5412 if not keep: 

5413 return 

5414 for col in keep: 

5415 if col in source.columns and col not in skip and col not in df.columns: 

5416 df[col] = source[col].to_numpy() 

5417 

5418 

5419def categorize_columns(raw: pd.DataFrame, schema: dict, registry: list) -> dict: 

5420 """Split a raw frame's columns into {mapped, detected_optional, unclaimed}. 

5421 

5422 ``mapped`` = source columns the schema references; ``detected_optional`` = 

5423 registry entries present in the frame (each ``{source, dest, category}``); 

5424 ``unclaimed`` = everything else (offered as filter fields / extra keeps). 

5425 

5426 AN-32: a registry column mapped as a *reading measure* is not also a 

5427 detected extra — `IA_DWELL_TIME` mapped as TFD was offered, pre-kept, as 

5428 `total_fixation_duration_ms` — and a source the registry lists twice (a 

5429 compatibility alias) is detected once, under its first entry. Only the 

5430 measure mapping claims a column here: a registry column the Trial ID is 

5431 composed from (`repeated_reading_trial`) is still a detected trial 

5432 condition, which is what offers it as a trial filter.""" 

5433 mapped = {c for c in _schema_source_columns(schema) if c in raw.columns} 

5434 as_measure = {schema.get(key) for key in READING_MEASURE_KEYS if schema.get(key)} 

5435 detected: list = [] 

5436 seen: set = set() 

5437 for src, dest, _kind, category in registry: 

5438 if src in raw.columns and src not in as_measure and src not in seen: 

5439 seen.add(src) 

5440 detected.append({"source": src, "dest": dest, "category": category}) 

5441 detected_sources = {d["source"] for d in detected} 

5442 unclaimed = [ 

5443 c for c in raw.columns if c not in mapped and c not in detected_sources 

5444 ] 

5445 return {"mapped": mapped, "detected_optional": detected, "unclaimed": unclaimed} 

5446 

5447 

5448def compute_keep_columns( 

5449 schema: dict, 

5450 *, 

5451 optional_sources: Iterable[str] | None = None, 

5452 filter_fields: Iterable[str] | None = None, 

5453 keep_columns: Iterable[str] | None = None, 

5454) -> set: 

5455 """Source columns to retain before normalization (everything else is dropped 

5456 for speed). Union of: schema-mapped sources, always-kept structural columns, 

5457 chosen optional fields, chosen filter fields, and extra keep columns.""" 

5458 keep = set(_schema_source_columns(schema)) 

5459 # Structural columns consulted directly by normalize_* (not via schema). 

5460 for col in ( 

5461 SOURCE_FILE_COLUMN, 

5462 "unique_trial_id", 

5463 "unique_paragraph_id", 

5464 "TRIAL_INDEX", 

5465 "trial_index", 

5466 ): 

5467 keep.add(col) 

5468 for group in (optional_sources, filter_fields, keep_columns): 

5469 if group: 

5470 keep.update(group) 

5471 return keep 

5472 

5473 

5474def _copy_screen_fields( 

5475 df: pd.DataFrame, source: pd.DataFrame, schema: dict 

5476) -> pd.DataFrame: 

5477 """Copy mapped part identity/metadata and normalize it in one place.""" 

5478 fields = ( 

5479 ("screen_id", SCREEN_ID, False), 

5480 ("screen_index", SCREEN_INDEX, True), 

5481 ("screen_timestamp", SCREEN_TIMESTAMP, True), 

5482 ("screen_fixation_id", SCREEN_FIXATION_ID, False), 

5483 ("canvas_width", CANVAS_WIDTH, True), 

5484 ("canvas_height", CANVAS_HEIGHT, True), 

5485 ) 

5486 for schema_key, destination, numeric in fields: 

5487 column = schema.get(schema_key) 

5488 if not column: 

5489 continue 

5490 values = source[column] 

5491 df[destination] = _to_number(values) if numeric else values 

5492 # BUG-79: UX-88 took `screen_index` out of the mapping on the premise that 

5493 # the public corpora stamp it onto their frames — but this function rebuilds 

5494 # the frame from the mapping, so the stamp was dropped and screen order 

5495 # re-derived from row order: AOI-file order on the words, each reader's 

5496 # onset order on the fixations. MultiplEYE's per-reader question order then 

5497 # conflicted and the 🗂️ Data page crashed. A canonical column rides through 

5498 # — but only onto a frame the mapping made multipart (DATA-59). With no 

5499 # screen field mapped, a raw `screen_index` column is just a column: riding 

5500 # it through derived a `screen_id` from it, so clearing the screen fields 

5501 # in the mapping still made the AOI table multipart while the fixations 

5502 # were not, and the pair was refused. 

5503 if ( 

5504 SCREEN_ID in df.columns 

5505 and SCREEN_INDEX not in df.columns 

5506 and SCREEN_INDEX in source.columns 

5507 ): 

5508 df[SCREEN_INDEX] = _to_number(source[SCREEN_INDEX]) 

5509 return normalize_screen_identity(df) 

5510 

5511 

5512def _text_id_mapped_flag( 

5513 source: pd.DataFrame, schema: dict, *, renormalizing: bool 

5514) -> bool | np.ndarray | None: 

5515 """What ``normalize_*`` write to ``_text_id_mapped`` (``None``: nothing). 

5516 

5517 A schema Text ID — auto-detected or picked — or a literal 

5518 ``unique_paragraph_id`` is mapped. On ✏️ Edit dataset the proposal maps 

5519 Text ID to the stored ``text_id`` column itself, which says nothing new: the 

5520 stored frame's own flag is kept, and a frame stored without one gets none 

5521 (so its values decide, as before).""" 

5522 text = schema.get("text_id") 

5523 if renormalizing and text and trial_mapping_columns(text) == ["text_id"]: 

5524 if TEXT_ID_MAPPED in source.columns: 

5525 return source[TEXT_ID_MAPPED].fillna(False).astype(bool).to_numpy() 

5526 return None 

5527 return bool(text) or "unique_paragraph_id" in source.columns 

5528 

5529 

5530def _drop_reserved_columns( 

5531 raw: pd.DataFrame, schema: dict, *, table: str 

5532) -> pd.DataFrame: 

5533 """Drop incoming columns named like :data:`INTERNAL_COLUMNS`, quietly. 

5534 

5535 The pipeline reads those names as its own bookkeeping — a join route, the 

5536 stimulus flag, the Edit-dataset collapse key — so an incoming column that 

5537 carries one must not reach any of them (DATA-49); normalization rebuilds 

5538 them. The names are the pipeline's own, so the usual source is a table 

5539 Scanpath Studio wrote being loaded again, a normal round-trip: logged at 

5540 INFO, not warned about. One the mapping names is the user's data and is 

5541 left alone.""" 

5542 referenced = _schema_source_columns(schema) 

5543 clash = sorted( 

5544 c for c in INTERNAL_COLUMNS if c in raw.columns and c not in referenced 

5545 ) 

5546 if not clash: 

5547 return raw 

5548 _LOGGER.info( 

5549 "%s: dropped %s, Scanpath Studio's own bookkeeping column(s); they are " 

5550 "rebuilt on load.", 

5551 table, 

5552 ", ".join(repr(c) for c in clash), 

5553 ) 

5554 return raw.drop(columns=clash) 

5555 

5556 

5557def normalize_words( 

5558 words: pd.DataFrame, 

5559 schema: dict[str, str], 

5560 *, 

5561 keep_columns: set | None = None, 

5562 _renormalizing: bool = False, 

5563) -> pd.DataFrame: 

5564 if not _renormalizing: 

5565 words = _drop_reserved_columns(words, schema, table="Words table") 

5566 _warn_normalization_issues(words, schema, table="Words table") 

5567 words = _drop_rows_missing_identity(words, schema) 

5568 # The explicit index makes scalar assignments (e.g. the stimulus-level 

5569 # participant placeholder) fill every row even when assigned first. 

5570 df = pd.DataFrame(index=words.index) 

5571 if schema.get("participant"): 

5572 # str or list (a composite participant id, joined like the trial id). 

5573 df["participant_id"] = trial_id_series(words, schema["participant"]) 

5574 else: 

5575 # Stimulus-level word/AoI table (one row per word per text, shared by 

5576 # all participants) — broadcast_stimulus_words() expands it across the 

5577 # participants found in the fixations. 

5578 df["participant_id"] = STIMULUS_PARTICIPANT 

5579 df[STIMULUS_WORDS_FLAG] = True 

5580 trial_cols = trial_mapping_columns(schema["trial"]) 

5581 if len(trial_cols) > 1: 

5582 # User-composed unique trial ID: authoritative, so it wins over a raw 

5583 # `unique_trial_id` column and needs no repeated-reading suffixing. 

5584 df["trial_id"] = trial_id_series(words, trial_cols) 

5585 df["unique_trial_id"] = df["trial_id"] 

5586 unsuffixed = df["trial_id"] 

5587 else: 

5588 # The mapped column, always (BUG-58). A literal `unique_trial_id` 

5589 # column used to win over whatever the mapping named, so a Trial ID 

5590 # picked by hand was silently replaced on any table that carried one — 

5591 # and a pair where only one side did joined on nothing. Auto-detection 

5592 # proposes `unique_trial_id` first, so it is still used by default. 

5593 trial_col = trial_cols[0] 

5594 df["trial_id"] = stable_id(words[trial_col]) 

5595 # The id before any repeat suffix names the text that was read. 

5596 unsuffixed = df["trial_id"] 

5597 if schema.get("participant"): 

5598 df = _disambiguate_repeated_readings(df, words, trial_col) 

5599 if "unique_trial_id" in words.columns: 

5600 # The mapped id *is* the unique trial id (BUG-58) — never the raw 

5601 # column's own values, which the trial picker would otherwise key 

5602 # on (`utils.build_combo_options` prefers `unique_trial_id`). 

5603 df["unique_trial_id"] = df["trial_id"] 

5604 if "unique_paragraph_id" in words.columns: 

5605 df["unique_text_id"] = stable_id(words["unique_paragraph_id"]) 

5606 df["text_id"] = df["unique_text_id"] 

5607 elif schema.get("text_id"): 

5608 # str or list (a composite text id, joined like the trial id). 

5609 df["text_id"] = trial_id_series(words, schema["text_id"]) 

5610 else: 

5611 # DATA-49: a repeated reading's text is the id it was suffixed from — 

5612 # the text a stimulus-level AOI table knows it by. 

5613 df["text_id"] = unsuffixed 

5614 mapped = _text_id_mapped_flag(words, schema, renormalizing=_renormalizing) 

5615 if mapped is not None: 

5616 df[TEXT_ID_MAPPED] = mapped 

5617 df = _copy_screen_fields(df, words, schema) 

5618 df["word_id"] = _to_number(words[schema["word_id"]]) 

5619 if schema.get("text"): 

5620 # BUG-53: a missing cell is an empty word, never NaN — pandas 3's 

5621 # `astype(str)` keeps NaN as NaN, and every " ".join over a trial's text 

5622 # downstream then raises on the float. 

5623 text = words[schema["text"]] 

5624 df["text"] = text.where(text.notna(), "").astype(str) 

5625 else: 

5626 df["text"] = df["word_id"].apply(lambda v: f"w{int(v)}" if pd.notna(v) else "") 

5627 df["text"] = df["text"].str.replace(r"\s+", " ", regex=True).str.strip() 

5628 if schema.get("line"): 

5629 df["line_idx"] = _to_number(words[schema["line"]]) 

5630 else: 

5631 df["line_idx"] = 1 

5632 

5633 if all(schema.get(k) for k in ["x", "y", "width", "height"]): 

5634 df["x"] = _to_number(words[schema["x"]]) 

5635 df["y"] = _to_number(words[schema["y"]]) 

5636 df["width"] = _to_number(words[schema["width"]]) 

5637 df["height"] = _to_number(words[schema["height"]]) 

5638 else: 

5639 left = _to_number(words[schema["left"]]) 

5640 right = _to_number(words[schema["right"]]) 

5641 top = _to_number(words[schema["top"]]) 

5642 bottom = _to_number(words[schema["bottom"]]) 

5643 df["x"] = left 

5644 df["y"] = top 

5645 df["width"] = right - left 

5646 df["height"] = bottom - top 

5647 

5648 emitted = _apply_optional_fields(df, words, WORD_OPTIONAL_FIELDS, keep_columns) 

5649 if keep_columns is not None: 

5650 _carry_extra_columns( 

5651 df, words, keep_columns, _schema_source_columns(schema) | emitted 

5652 ) 

5653 # AN-32: after the passthrough and the extras, so the mapping has the last 

5654 # word on every measure it names. 

5655 _apply_reading_measures(df, words, schema) 

5656 _blank_unfixated_measures(df) 

5657 

5658 df = _preserve_composite_columns(df, words, schema["trial"]) 

5659 return df 

5660 

5661 

5662def normalize_fixations( 

5663 fixations: pd.DataFrame, 

5664 schema: dict[str, str], 

5665 *, 

5666 keep_columns: set | None = None, 

5667 _renormalizing: bool = False, 

5668) -> pd.DataFrame: 

5669 if not _renormalizing: 

5670 fixations = _drop_reserved_columns(fixations, schema, table="Fixations") 

5671 _warn_normalization_issues(fixations, schema, table="Fixations", fixations=True) 

5672 fixations = _drop_rows_missing_identity(fixations, schema) 

5673 # Explicit index so a constant participant placeholder fills every row. 

5674 df = pd.DataFrame(index=fixations.index) 

5675 if schema.get("participant"): 

5676 # str or list (a composite participant id, joined like the trial id). 

5677 df["participant_id"] = trial_id_series(fixations, schema["participant"]) 

5678 else: 

5679 # No participant column → a single anonymous reader. 

5680 df["participant_id"] = SYNTHETIC_PARTICIPANT 

5681 trial_cols = trial_mapping_columns(schema["trial"]) 

5682 if len(trial_cols) > 1: 

5683 # User-composed unique trial ID — see normalize_words. 

5684 df["trial_id"] = trial_id_series(fixations, trial_cols) 

5685 df["unique_trial_id"] = df["trial_id"] 

5686 unsuffixed = df["trial_id"] 

5687 else: 

5688 trial_col = trial_cols[0] # the mapped column (BUG-58) 

5689 df["trial_id"] = stable_id(fixations[trial_col]) 

5690 # The id before any repeat suffix names the text that was read. 

5691 unsuffixed = df["trial_id"] 

5692 if schema.get("participant"): 

5693 df = _disambiguate_repeated_readings( 

5694 df, fixations, trial_col, record_base=True 

5695 ) 

5696 if "unique_trial_id" in fixations.columns: 

5697 # The mapped id *is* the unique trial id (BUG-58) — never the raw 

5698 # column's own values, which the trial picker would otherwise key 

5699 # on (`utils.build_combo_options` prefers `unique_trial_id`). 

5700 df["unique_trial_id"] = df["trial_id"] 

5701 if "unique_paragraph_id" in fixations.columns: 

5702 df["text_id"] = stable_id(fixations["unique_paragraph_id"]) 

5703 elif schema.get("text_id"): 

5704 # str or list (a composite text id, joined like the trial id). 

5705 df["text_id"] = trial_id_series(fixations, schema["text_id"]) 

5706 else: 

5707 # DATA-49: the text a repeated reading is of, without its `_r2`. 

5708 df["text_id"] = unsuffixed 

5709 mapped = _text_id_mapped_flag(fixations, schema, renormalizing=_renormalizing) 

5710 if mapped is not None: 

5711 df[TEXT_ID_MAPPED] = mapped 

5712 if "unique_paragraph_id" in fixations.columns: 

5713 df["unique_text_id"] = stable_id(fixations["unique_paragraph_id"]) 

5714 df = _copy_screen_fields(df, fixations, schema) 

5715 # X/Y may be unmapped for AOI-sequence datasets (no pixel coordinates) — 

5716 # left NaN here and filled from word-box centers by harmonize_frames(). 

5717 for coord in ("x", "y"): 

5718 if schema.get(coord): 

5719 df[coord] = _to_number(fixations[schema[coord]]) 

5720 else: 

5721 df[coord] = np.nan 

5722 # An unreadable duration / onset still falls back to 0, but no longer 

5723 # silently: `_warn_numeric_issues` below names the column (BUG-54). 

5724 # DATA-40: a duration / onset in seconds (Gazepoint), microseconds (Tobii) 

5725 # or nanoseconds (Pupil Labs Neon) is read in milliseconds. 

5726 duration = schema["duration"] 

5727 df["duration_ms"] = _as_ms(_to_number(fixations[duration]), duration).fillna(0) 

5728 

5729 if schema.get("timestamp"): 

5730 onset = schema["timestamp"] 

5731 df["timestamp_ms"] = _as_ms(_to_number(fixations[onset]), onset).fillna(0) 

5732 else: 

5733 df["timestamp_ms"] = df.groupby(list(PARENT_KEY), sort=False).cumcount() 

5734 

5735 if schema.get("fixation_id"): 

5736 df["fixation_id"] = fixations[schema["fixation_id"]] 

5737 else: 

5738 df["fixation_id"] = df.groupby(list(PARENT_KEY), sort=False).cumcount().add(1) 

5739 

5740 if SCREEN_ID in df.columns: 

5741 part_keys = grouping_columns(df) 

5742 if SCREEN_TIMESTAMP not in df.columns: 

5743 df[SCREEN_TIMESTAMP] = df.groupby(part_keys, sort=False).cumcount() 

5744 if SCREEN_FIXATION_ID not in df.columns: 

5745 df[SCREEN_FIXATION_ID] = df.groupby(part_keys, sort=False).cumcount().add(1) 

5746 

5747 if schema.get("word_id"): 

5748 df["word_id"] = _to_number(fixations[schema["word_id"]]) 

5749 else: 

5750 df["word_id"] = np.nan 

5751 

5752 # pass_index / saccade_type / saccade_amplitude / eye are no longer schema 

5753 # fields — they ride through _apply_optional_fields (FIX_OPTIONAL_FIELDS) when 

5754 # the data carries them. saccade_amplitude is also recomputed from X/Y by 

5755 # measures.enrich_fixations when absent. 

5756 emitted = _apply_optional_fields(df, fixations, FIX_OPTIONAL_FIELDS, keep_columns) 

5757 if keep_columns is not None: 

5758 _carry_extra_columns( 

5759 df, fixations, keep_columns, _schema_source_columns(schema) | emitted 

5760 ) 

5761 

5762 df = _preserve_composite_columns(df, fixations, schema["trial"]) 

5763 synthesized = _synthesized_timestamps(fixations, schema, _renormalizing) 

5764 if synthesized is None: 

5765 df = df.drop(columns=[TIMESTAMP_SYNTHESIZED], errors="ignore") 

5766 else: 

5767 df[TIMESTAMP_SYNTHESIZED] = synthesized 

5768 

5769 df["order_in_trial"] = ( 

5770 df.sort_values(["timestamp_ms", "duration_ms"]) 

5771 .groupby(list(PARENT_KEY), sort=False) 

5772 .cumcount() 

5773 + 1 

5774 ) 

5775 if SCREEN_ID in df.columns: 

5776 df["order_in_screen"] = ( 

5777 df.sort_values([SCREEN_INDEX, SCREEN_TIMESTAMP, "duration_ms"]) 

5778 .groupby(grouping_columns(df), sort=False) 

5779 .cumcount() 

5780 + 1 

5781 ) 

5782 return df 

5783 

5784 

5785# Canonical columns produced by normalize_words / normalize_fixations. Used to 

5786# build typed empty frames when a dataset ships only one of the two reports, 

5787# so every downstream consumer can keep selecting columns unconditionally. 

5788WORDS_CANONICAL_COLUMNS: dict[str, str] = { 

5789 "participant_id": "object", 

5790 "trial_id": "object", 

5791 "text_id": "object", 

5792 "word_id": "float64", 

5793 "text": "object", 

5794 "line_idx": "float64", 

5795 "x": "float64", 

5796 "y": "float64", 

5797 "width": "float64", 

5798 "height": "float64", 

5799} 

5800FIX_CANONICAL_COLUMNS: dict[str, str] = { 

5801 "participant_id": "object", 

5802 "trial_id": "object", 

5803 "text_id": "object", 

5804 "x": "float64", 

5805 "y": "float64", 

5806 "duration_ms": "float64", 

5807 "timestamp_ms": "float64", 

5808 "fixation_id": "float64", 

5809 "word_id": "float64", 

5810 "order_in_trial": "int64", 

5811} 

5812 

5813 

5814def empty_words_frame() -> pd.DataFrame: 

5815 """An empty words frame with the canonical post-normalization columns.""" 

5816 return pd.DataFrame( 

5817 {col: pd.Series(dtype=dt) for col, dt in WORDS_CANONICAL_COLUMNS.items()} 

5818 ) 

5819 

5820 

5821def empty_fixations_frame() -> pd.DataFrame: 

5822 """An empty fixations frame with the canonical post-normalization columns.""" 

5823 return pd.DataFrame( 

5824 {col: pd.Series(dtype=dt) for col, dt in FIX_CANONICAL_COLUMNS.items()} 

5825 ) 

5826 

5827 

5828# Identity columns recomputed (not carried) by a remap — see remap_normalized_frame. 

5829_REMAP_DERIVED_IDS = ("unique_trial_id", "unique_text_id", "unique_paragraph_id") 

5830 

5831 

5832def repeat_bases(fixations: pd.DataFrame | None) -> dict: 

5833 """Each repeated reading's ``(reader, trial id)`` → the trial id it was 

5834 recorded under. 

5835 

5836 From ``_base_trial_id`` when the fixations carry it. A stored frame from 

5837 before DATA-49 does not, so its suffixes are read back instead: a trial 

5838 ``X_rN`` of a reader who also has a trial ``X`` is the repeat 

5839 ``_disambiguate_repeated_readings`` made of it. Only the pre-provenance 

5840 Edit-dataset path uses the second form, to fold a stored repeat's copy of 

5841 the boxes back into the AOI trial it copied.""" 

5842 if fixations is None or fixations.empty or "trial_id" not in fixations: 

5843 return {} 

5844 if "participant_id" not in fixations.columns: 

5845 return {} 

5846 if BASE_TRIAL_ID in fixations.columns: 

5847 pairs = fixations[["participant_id", "trial_id", BASE_TRIAL_ID]] 

5848 pairs = pairs.dropna().drop_duplicates().astype(str) 

5849 pairs = pairs[pairs["trial_id"] != pairs[BASE_TRIAL_ID]] 

5850 return dict( 

5851 zip( 

5852 zip(pairs["participant_id"], pairs["trial_id"]), 

5853 pairs[BASE_TRIAL_ID], 

5854 ) 

5855 ) 

5856 keys = fixations[["participant_id", "trial_id"]].dropna().drop_duplicates() 

5857 keys = keys.astype(str) 

5858 base = keys["trial_id"].str.extract(_REPEAT_SUFFIX, expand=False) 

5859 candidates = keys.assign(base=base).dropna(subset=["base"]) 

5860 have = pd.MultiIndex.from_frame(keys[["participant_id", "trial_id"]]) 

5861 known = pd.MultiIndex.from_arrays( 

5862 [candidates["participant_id"], candidates["base"]] 

5863 ).isin(have) 

5864 chosen = candidates[known] 

5865 return dict(zip(zip(chosen["participant_id"], chosen["trial_id"]), chosen["base"])) 

5866 

5867 

5868#: A repeated reading's suffix exactly as `_disambiguate_repeated_readings` 

5869#: writes it: `_r2`, `_r3` …, never `_r0`/`_r1` or a zero-padded number. 

5870_REPEAT_SUFFIX = r"^(.+)_r(?:[2-9]|[1-9]\d+)$" 

5871 

5872 

5873def _map_repeats(frame: pd.DataFrame, repeat_of: dict) -> pd.Series: 

5874 """``frame``'s trial ids with each repeat replaced by its base, looked up 

5875 per ``(reader, trial)`` in a :func:`repeat_bases` mapping.""" 

5876 ids = frame["trial_id"].astype(str) 

5877 if not repeat_of or "participant_id" not in frame.columns: 

5878 return ids 

5879 table = pd.Series( 

5880 list(repeat_of.values()), index=pd.MultiIndex.from_tuples(list(repeat_of)) 

5881 ) 

5882 wanted = pd.MultiIndex.from_arrays([frame["participant_id"].astype(str), ids]) 

5883 found = table.reindex(wanted).to_numpy() 

5884 return pd.Series(np.where(pd.isna(found), ids.to_numpy(), found), index=frame.index) 

5885 

5886 

5887def _synthesized_timestamps( 

5888 fixations: pd.DataFrame, schema: dict, renormalizing: bool 

5889) -> pd.Series | None: 

5890 """The :data:`TIMESTAMP_SYNTHESIZED` flags for a normalized fixations frame. 

5891 

5892 Every row when no onset is mapped. A remap that keeps the stored 

5893 ``timestamp_ms`` as the onset keeps the stored flags — those numbers are 

5894 still the ones normalization made up. ``None`` when the timestamps are 

5895 the data's own.""" 

5896 onset = schema.get("timestamp") 

5897 if not onset: 

5898 return pd.Series(True, index=fixations.index) 

5899 if ( 

5900 renormalizing 

5901 and onset == "timestamp_ms" 

5902 and TIMESTAMP_SYNTHESIZED in fixations.columns 

5903 ): 

5904 return fixations[TIMESTAMP_SYNTHESIZED].fillna(False).astype(bool) 

5905 return None 

5906 

5907 

5908def remap_normalized_frame( 

5909 frame: pd.DataFrame, 

5910 schema: dict[str, str | None], 

5911 *, 

5912 kind: str, 

5913 repeat_of: dict | None = None, 

5914) -> pd.DataFrame: 

5915 """Re-derive an already-normalized frame under a new column mapping. 

5916 

5917 Stored datasets keep only their post-normalization frames — canonical column 

5918 names (``duration_ms``, ``x``, ``word_id``, …) plus any kept extras; the 

5919 original upload columns are gone. To change the mapping without re-uploading, 

5920 re-run the matching ``normalize_*`` over the *normalized* frame, treating its 

5921 current columns as the source universe. ``schema`` therefore references 

5922 canonical/extra column names (e.g. ``{"duration": "duration_ms", 

5923 "trial": "trial_id", ...}``). 

5924 

5925 The precomputed identity columns (``unique_trial_id`` etc.) are dropped first 

5926 so the new Trial/Text mapping is authoritative: otherwise ``normalize_*`` 

5927 would keep deriving ``trial_id`` from the existing ``unique_trial_id`` and 

5928 silently ignore a changed Trial ID pick. A derived-id column the new schema 

5929 *references* is NOT dropped, though — a composite trial id can be built from 

5930 ``unique_paragraph_id``, and dropping a chosen component would make 

5931 ``trial_id_series`` raise ``KeyError``. Every surviving column is kept 

5932 (``keep_columns`` = all current columns) so the remap only reassigns roles 

5933 and never drops data that already survived the first normalization. After 

5934 normalization ``unique_trial_id`` / ``unique_text_id`` are restored 

5935 (= ``trial_id`` / ``text_id``) when the single-column path didn't set them, 

5936 so the frame's identity columns stay consistent with the composite path and 

5937 downstream readers of ``unique_text_id`` keep working. 

5938 

5939 DATA-39: a **words** frame remapped with no Participant is a stimulus-level 

5940 AOI table, exactly as it was at import — ``normalize_words`` re-flags it for 

5941 ``broadcast_stimulus_words``. But the stored frame was *already* broadcast 

5942 (one copy of every trial's words per reader), so only the first reader's 

5943 copy of each trial is kept here, and the caller must run 

5944 ``harmonize_frames`` to broadcast it again. Skipping either step is the bug 

5945 this fixes: without the collapse every reader gets every reader's boxes; 

5946 without the harmonize every word is left on the ``""`` placeholder reader, 

5947 so no trial finds its boxes and the scanpath plot loses its AOIs and its 

5948 text. The collapse picks a *reader*, never a key: deduplicating on 

5949 ``word_id`` would also merge rows that are not copies at all — character 

5950 AOIs sharing a word id, or ids that do not parse as numbers and all fold to 

5951 NaN.""" 

5952 referenced = _schema_source_columns(schema) 

5953 derived = list(_REMAP_DERIVED_IDS) 

5954 if trial_mapping_columns(schema.get("trial") or []) != ["trial_id"]: 

5955 # A repeat's recorded id belongs to the trial mapping that suffixed it: 

5956 # kept while the Trial ID stays the stored `trial_id` (the editor's own 

5957 # proposal), re-derived by the disambiguation under any other pick. 

5958 derived.append(BASE_TRIAL_ID) 

5959 working = frame.drop( 

5960 columns=[c for c in derived if c in frame.columns and c not in referenced] 

5961 ) 

5962 if ( 

5963 kind == "fixations" 

5964 and repeat_of 

5965 and BASE_TRIAL_ID not in working.columns 

5966 and BASE_TRIAL_ID not in derived 

5967 ): 

5968 # A fixations frame stored before `_base_trial_id` existed: give its 

5969 # repeats the id they were recorded under, so the join still finds 

5970 # their boxes after the save. 

5971 bases = _map_repeats(working, repeat_of) 

5972 if (bases != working["trial_id"].astype(str)).any(): 

5973 working = working.assign(**{BASE_TRIAL_ID: bases}) 

5974 if ( 

5975 kind == "words" 

5976 and not schema.get("participant") 

5977 and "participant_id" in working.columns 

5978 and "trial_id" in working.columns 

5979 and not working.empty 

5980 ): 

5981 if AOI_TRIAL_ID in working.columns: 

5982 # Broadcast here (DATA-49): each copy names the AOI trial it came 

5983 # from, whichever route it took — so one reading's copy of each AOI 

5984 # trial, stamped back with that trial's own id, *is* the stimulus 

5985 # table again. No guessing from content, no suffixed repeats left. 

5986 copy_keys = [AOI_TRIAL_ID, *[SCREEN_ID] * (SCREEN_ID in working.columns)] 

5987 reading = ( 

5988 working["participant_id"].astype(str) 

5989 + "\x1f" 

5990 + working["trial_id"].astype(str) 

5991 ) 

5992 else: 

5993 # Stored before the provenance column: joined by trial id, so the 

5994 # readers of a stimulus share its trial id — except a repeat, whose 

5995 # copy carries its `_r2` id. Fold it back into the trial it copied 

5996 # (`repeat_of`, from the fixations), or it would become a phantom 

5997 # AOI trial of its own and make its text ambiguous for good. 

5998 # One reading's copy — its reader *and* its own (stored) trial id, 

5999 # so a reader's repeat is not kept as a second copy. 

6000 reading = ( 

6001 working["participant_id"].astype(str) 

6002 + "\x1f" 

6003 + working["trial_id"].astype(str) 

6004 ) 

6005 if repeat_of: 

6006 working = working.assign(trial_id=_map_repeats(working, repeat_of)) 

6007 copy_keys = [c for c in ("trial_id", SCREEN_ID) if c in working.columns] 

6008 first = reading.groupby( 

6009 [working[c] for c in copy_keys], dropna=False, sort=False 

6010 ).transform("first") 

6011 working = working[reading == first] 

6012 if AOI_TRIAL_ID in working.columns: 

6013 working = working.assign(trial_id=working[AOI_TRIAL_ID]).drop( 

6014 columns=[AOI_TRIAL_ID] 

6015 ) 

6016 keep = set(working.columns) 

6017 if kind == "words": 

6018 result = normalize_words( 

6019 working, schema, keep_columns=keep, _renormalizing=True 

6020 ) 

6021 elif kind == "fixations": 

6022 result = normalize_fixations( 

6023 working, schema, keep_columns=keep, _renormalizing=True 

6024 ) 

6025 elif kind == "raw_gaze": 

6026 result = normalize_raw_gaze(working, schema, keep_columns=keep) 

6027 else: 

6028 raise ValueError(f"unknown frame kind: {kind!r}") 

6029 if "unique_trial_id" not in result.columns and "trial_id" in result.columns: 

6030 result["unique_trial_id"] = result["trial_id"] 

6031 if "unique_text_id" not in result.columns and "text_id" in result.columns: 

6032 result["unique_text_id"] = result["text_id"] 

6033 return result 

6034 

6035 

6036def _union_column_values( 

6037 words: pd.DataFrame, fixations: pd.DataFrame, column: str 

6038) -> list: 

6039 """Sorted union of a column's values across both frames (either may be 

6040 empty — single-report datasets have words or fixations, not both).""" 

6041 values: set = set() 

6042 for df in (words, fixations): 

6043 if df is not None and not df.empty and column in df.columns: 

6044 values.update(df[column].unique()) 

6045 return sorted(values) 

6046 

6047 

6048def filter_data( 

6049 words: pd.DataFrame, 

6050 fixations: pd.DataFrame, 

6051 filters: dict, 

6052) -> tuple[pd.DataFrame, pd.DataFrame]: 

6053 # When the participant/trial selection covers the whole frame (the default — 

6054 # any narrowing already happened upstream in filter_trials), skip the two 

6055 # O(n) membership masks entirely; only the optional fixation-level filters 

6056 # below apply. ``default_filters`` sets the cover-all flags. 

6057 cover_all = bool( 

6058 filters.get("_participants_cover_all") and filters.get("_trials_cover_all") 

6059 ) 

6060 if cover_all: 

6061 # participant/trial cover the whole frame and default_filters set the 

6062 # pass/saccade/eye filters to their full value sets (no-ops), so nothing 

6063 # can narrow the fixations — return the frames untouched (no full-frame 

6064 # mask, no copy), the common large-upload case. 

6065 return words, fixations 

6066 

6067 # As in `filter_trials`: only a missing key means "every participant". An 

6068 # explicit empty list is a narrowing that matches nobody. 

6069 participants = filters.get("participants") 

6070 if participants is None: 

6071 participants = _union_column_values(words, fixations, "participant_id") 

6072 trials = filters.get("trials") or _union_column_values(words, fixations, "trial_id") 

6073 word_mask = words["participant_id"].isin(participants) & words["trial_id"].isin( 

6074 trials 

6075 ) 

6076 words_filtered = words[word_mask] 

6077 fix_mask = fixations["participant_id"].isin(participants) & fixations[ 

6078 "trial_id" 

6079 ].isin(trials) 

6080 if "pass_index" in fixations.columns: 

6081 pass_indices = filters.get("pass_indices") 

6082 if pass_indices: 

6083 fix_mask &= fixations["pass_index"].isin(pass_indices) 

6084 if "saccade_type" in fixations.columns: 

6085 saccade_types = filters.get("saccade_types") 

6086 if saccade_types: 

6087 fix_mask &= fixations["saccade_type"].isin(saccade_types) 

6088 if "eye" in fixations.columns: 

6089 eyes = filters.get("eyes") 

6090 if eyes: 

6091 fix_mask &= fixations["eye"].isin(eyes) 

6092 fixations_filtered = fixations[fix_mask] 

6093 return words_filtered, fixations_filtered 

6094 

6095 

6096def filter_trials( 

6097 words: pd.DataFrame, 

6098 fixations: pd.DataFrame, 

6099 participants: list | None = None, 

6100 metadata: dict[str, set] | None = None, 

6101 ranges: dict[str, tuple[float, float]] | None = None, 

6102 drop_unknown: Iterable[str] | None = None, 

6103) -> tuple[pd.DataFrame, pd.DataFrame]: 

6104 """Narrow words + fixations by participant and by trial metadata. 

6105 

6106 ``metadata`` maps a column name to the set of allowed values (membership). 

6107 Only columns present on a frame are applied, so a condition like 

6108 ``question_preview`` (Hunting/Gathering) narrows both words and fixations — 

6109 the column is copied onto both during normalization. A falsy selection means 

6110 "no constraint". 

6111 

6112 ``ranges`` (UX-49) is the *continuous* counterpart: column → ``(lo, hi)`` 

6113 inclusive bounds for a numeric trial-level column, where enumerating the 

6114 distinct values would be useless. **Rows with no value survive**: a range is 

6115 a narrowing control, so a reader missing a comprehension score is not what 

6116 the user asked to exclude — and pandas compares ``NaN`` as ``False``, so the 

6117 obvious bare ``.between()`` would silently drop every one of them. 

6118 ``drop_unknown`` names the ranged columns whose researcher unticked *Keep 

6119 unknown values*: there, a row with no value is left out with the rest. 

6120 """ 

6121 w, f = words, fixations 

6122 dropping = set(drop_unknown or ()) 

6123 # `None` is "no constraint"; an **empty list is a constraint that nothing 

6124 # satisfies** and must empty the pool. The two were conflated while every 

6125 # producer could only emit None-or-non-empty, but DATA-20's metadata 

6126 # narrowing intersects sets and can legitimately land on zero readers — 

6127 # under the old falsy test that silently applied no filter at all, so an 

6128 # impossible combination showed the *whole* corpus. 

6129 if participants is not None: 

6130 # participant_id is already string after normalization (as the metadata 

6131 # filters below also assume), so skip a full-column .astype(str) recast. 

6132 keep = set(map(str, participants)) 

6133 w = w[w["participant_id"].isin(keep)] 

6134 f = f[f["participant_id"].isin(keep)] 

6135 for col, allowed in (metadata or {}).items(): 

6136 if not allowed: 

6137 continue 

6138 allowed = set(allowed) 

6139 if col in w.columns: 

6140 w = w[w[col].isin(allowed)] 

6141 if col in f.columns: 

6142 f = f[f[col].isin(allowed)] 

6143 for col, bounds in (ranges or {}).items(): 

6144 if not bounds: 

6145 continue 

6146 lo, hi = bounds 

6147 for frame_name in ("w", "f"): 

6148 frame = w if frame_name == "w" else f 

6149 if col not in frame.columns: 

6150 continue 

6151 values = pd.to_numeric(frame[col], errors="coerce") 

6152 mask = values.between(lo, hi) 

6153 if col not in dropping: 

6154 mask |= values.isna() 

6155 if frame_name == "w": 

6156 w = w[mask] 

6157 else: 

6158 f = f[mask] 

6159 return w, f 

6160 

6161 

6162def trial_keys(frame: pd.DataFrame) -> set: 

6163 """The distinct ``(participant_id, trial_id)`` string keys a frame carries. 

6164 

6165 Deduplicates before materializing the tuples, so it stays cheap on a 

6166 sample-level table (raw gaze) where every trial spans thousands of rows. 

6167 A missing/empty frame, or one without both id columns, yields an empty set. 

6168 """ 

6169 if frame is None or frame.empty: 

6170 return set() 

6171 if "participant_id" not in frame.columns or "trial_id" not in frame.columns: 

6172 return set() 

6173 pairs = frame[["participant_id", "trial_id"]].drop_duplicates() 

6174 return { 

6175 (str(p), str(t)) for p, t in zip(pairs["participant_id"], pairs["trial_id"]) 

6176 } 

6177 

6178 

6179def text_ids(*frames: pd.DataFrame | None) -> set[str]: 

6180 """The distinct text ids across every frame that carries one (DATA-50). 

6181 

6182 ``unique_text_id`` when any frame has it, else ``text_id`` — one id space, 

6183 never a union of the two. Counting the words table alone said **0 texts** 

6184 for a fixations-only dataset whose fixations name twelve. 

6185 """ 

6186 present = [f for f in frames if f is not None and not f.empty] 

6187 column = ( 

6188 "unique_text_id" 

6189 if any("unique_text_id" in f.columns for f in present) 

6190 else "text_id" 

6191 ) 

6192 found: set[str] = set() 

6193 for frame in present: 

6194 if column in frame.columns: 

6195 found.update(str(v) for v in frame[column].dropna().unique()) 

6196 return found 

6197 

6198 

6199def filter_frame_to_keys(frame: pd.DataFrame, keys: set) -> pd.DataFrame: 

6200 """Keep only rows whose ``(participant_id, trial_id)`` is in ``keys``. 

6201 

6202 Single-frame counterpart of :func:`filter_to_keys`. BUG-12: the raw-gaze 

6203 samples table has to be narrowed by the annotation filters (favorites / 

6204 tags) exactly like the words and fixations frames, or a sample row for an 

6205 unstarred trial survives "⭐ Favorites only". Vectorized via a MultiIndex 

6206 membership test so it stays fast on large tables. 

6207 """ 

6208 if frame is None or frame.empty: 

6209 return frame 

6210 if "participant_id" not in frame.columns or "trial_id" not in frame.columns: 

6211 return frame 

6212 idx = pd.MultiIndex.from_arrays( 

6213 [frame["participant_id"].astype(str), frame["trial_id"].astype(str)] 

6214 ) 

6215 return frame[idx.isin(keys)] 

6216 

6217 

6218def filter_to_keys( 

6219 words: pd.DataFrame, 

6220 fixations: pd.DataFrame, 

6221 keys: set, 

6222) -> tuple[pd.DataFrame, pd.DataFrame]: 

6223 """Keep only rows whose (participant_id, trial_id) is in ``keys``. 

6224 

6225 ``keys`` is a set of ``(str, str)`` tuples. Used to apply annotation-based 

6226 filtering (favorites / tags). Vectorized via a MultiIndex membership test so 

6227 it stays fast on large fixation tables.""" 

6228 return filter_frame_to_keys(words, keys), filter_frame_to_keys(fixations, keys) 

6229 

6230 

6231def raw_gaze_in_pool( 

6232 raw_gaze: pd.DataFrame, 

6233 words_all: pd.DataFrame, 

6234 fixations_all: pd.DataFrame, 

6235 words_pool: pd.DataFrame, 

6236 fixations_pool: pd.DataFrame, 

6237) -> pd.DataFrame: 

6238 """The raw-gaze rows of the trials in the current pool (VIZ-45). 

6239 

6240 A trial the words or fixations table knows is in the pool when it survived 

6241 the filters there, so its samples follow it. A trial **only the raw gaze 

6242 knows** — a raw-gaze-only dataset's every trial, or the samples-only trials 

6243 of a dataset that has fixations for others — has nothing there to survive, 

6244 and used to be dropped for it: the old narrowing kept only the participants 

6245 and trials the other two tables listed. Those trials stay, narrowed only by 

6246 what applies to the samples themselves (participant, annotations, the 

6247 trial-metadata keys — applied to ``raw_gaze`` before this). 

6248 

6249 ``words_all`` / ``fixations_all`` are the frames *before* the filters, which 

6250 is what tells "filtered out" from "never there". Returns ``raw_gaze`` itself 

6251 when nothing is dropped — always when the pools are those very frames (no 

6252 filter set), with no scan at all — and otherwise the narrowed frame from a 

6253 `frame_cache`, so a rerun under the same filters reuses it rather than 

6254 re-masking every sample. 

6255 """ 

6256 if raw_gaze is None or raw_gaze.empty: 

6257 return pd.DataFrame() 

6258 if (words_all is None or words_all.empty) and ( 

6259 fixations_all is None or fixations_all.empty 

6260 ): 

6261 return raw_gaze 

6262 if words_pool is words_all and fixations_pool is fixations_all: 

6263 # Unfiltered: every raw-gaze trial is either pooled or unknown to them. 

6264 return raw_gaze 

6265 key = tuple( 

6266 frame_fingerprint(frame) 

6267 for frame in (raw_gaze, words_all, fixations_all, words_pool, fixations_pool) 

6268 ) 

6269 

6270 def _build() -> pd.DataFrame: 

6271 known = _known_trial_keys( 

6272 words_all, 

6273 fixations_all, 

6274 cache_key=(frame_fingerprint(words_all), frame_fingerprint(fixations_all)), 

6275 ) 

6276 # Per raw-gaze table, not per filter change: `frame_cache` keeps one 

6277 # entry per slot, so toggling a filter back rebuilds this — and the 

6278 # samples' own keys are the one full pass worth not repeating. 

6279 present = _raw_gaze_trial_keys(raw_gaze, cache_key=frame_fingerprint(raw_gaze)) 

6280 pooled = trial_keys(words_pool) | trial_keys(fixations_pool) 

6281 keep = {k for k in present if k in pooled or k not in known} 

6282 return raw_gaze if keep == present else filter_frame_to_keys(raw_gaze, keep) 

6283 

6284 return frame_cache("raw_gaze_pool", key, _build) 

6285 

6286 

6287@st.cache_data(show_spinner=False, max_entries=8) 

6288def _raw_gaze_trial_keys(_raw_gaze: pd.DataFrame, cache_key) -> set: 

6289 """`trial_keys` of one (narrowed) raw-gaze table, cached on its fingerprint.""" 

6290 progress.report() 

6291 return trial_keys(_raw_gaze) 

6292 

6293 

6294@st.cache_data(show_spinner=False, max_entries=8) 

6295def _known_trial_keys( 

6296 _words_all: pd.DataFrame, _fixations_all: pd.DataFrame, cache_key 

6297) -> set: 

6298 """The trials the unfiltered words and fixations know, once per dataset — 

6299 not once per filter change (`raw_gaze_in_pool`).""" 

6300 progress.report() 

6301 return trial_keys(_words_all) | trial_keys(_fixations_all) 

6302 

6303 

6304# --------------------------------------------------------------------------- 

6305# VAL-7 — is a "trial" actually more than one reading? 

6306# 

6307# The inverse of BUG-23. A Trial ID mapping that does not fully identify a 

6308# reading concatenates several readings into one `trial_id`, and nothing about 

6309# the result looks wrong: the figure renders as an ordinary scanpath with a lot 

6310# of regressions. Three independent signals catch it, and a fourth names the fix. 

6311# 

6312# THE KEY IS `(participant_id, trial_id, screen_id)` — the multipart scientific 

6313# grouping key, not `(participant_id, trial_id)`. A legitimate two-screen trial 

6314# restarts `word_id` per screen, so the words signal reports duplicates that are 

6315# correct data when grouped by trial alone. The two fixation signals are immune 

6316# either way, because multipart keeps `fixation_id` and `timestamp_ms` 

6317# parent-global by design — which is exactly why they are the right cross-check 

6318# rather than a redundant one. 

6319# --------------------------------------------------------------------------- 

6320 

6321#: Columns that should hold ONE value inside a reading, so a second value is 

6322#: evidence that two readings were merged — and the column name *is* the remedy 

6323#: ("add `article_id` to the Trial ID mapping"). Deliberately a fixed list of 

6324#: identity/condition fields rather than every column: a per-word or 

6325#: per-fixation column varies within a trial by design, and scanning them all 

6326#: would bury the signal in noise. 

6327_TRIAL_WITNESS_COLUMNS: tuple[str, ...] = ( 

6328 "TRIAL_INDEX", 

6329 "trial_index", 

6330 "trial_num", 

6331 "article_id", 

6332 "article_batch", 

6333 "article_title", 

6334 "difficulty_level", 

6335 "question", 

6336 "question_preview", 

6337 "repeated_reading_trial", 

6338 "selected_answer", 

6339 "is_correct", 

6340 "genre", 

6341 "session", 

6342 "is_practice", 

6343 "unique_paragraph_id", 

6344 "paragraph_id", 

6345 "unique_text_id", 

6346 "text_id", 

6347 SOURCE_FILE_COLUMN, 

6348) 

6349 

6350 

6351def trial_identity_key(frame: pd.DataFrame) -> list[str]: 

6352 """The columns that identify one *reading* in ``frame``. 

6353 

6354 ``screen_id`` joins the key when the frame is multipart, because a screen is 

6355 a coordinate space of its own and its ids restart. 

6356 """ 

6357 key = [c for c in ("participant_id", "trial_id") if c in frame.columns] 

6358 if SCREEN_ID in frame.columns: 

6359 key.append(SCREEN_ID) 

6360 return key 

6361 

6362 

6363def _sample_trials( 

6364 words: pd.DataFrame, fixations: pd.DataFrame, limit: int | None, seed: int 

6365) -> tuple[pd.DataFrame, pd.DataFrame, int | None]: 

6366 """Narrow both frames to ``limit`` trials, or leave them alone (VAL-7). 

6367 

6368 Samples the *trial keys* rather than rows, so a chosen trial arrives whole — 

6369 a half-read trial would look exactly like the two-readings-under-one-id 

6370 defect this screens for. Returns the frames plus the corpus size sampled 

6371 from, or ``None`` when no sampling happened. 

6372 """ 

6373 if not limit or limit <= 0: 

6374 return words, fixations, None 

6375 keys = trial_keys(words) | trial_keys(fixations) 

6376 if len(keys) <= limit: 

6377 return words, fixations, None 

6378 rng = np.random.default_rng(seed) 

6379 ordered = sorted(keys) # a set's order is not stable across processes 

6380 picked = {ordered[i] for i in rng.choice(len(ordered), size=limit, replace=False)} 

6381 kept_words, kept_fixations = filter_to_keys(words, fixations, picked) 

6382 return kept_words, kept_fixations, len(keys) 

6383 

6384 

6385#: Trials VAL-7's identity check examines by default before it starts sampling. 

6386#: Measured on the full OneStop corpus (24,046 trials, 5.0 M rows): the census 

6387#: takes 4.23 s on every dataset load, for one warning line, and this sample 

6388#: takes 1.15 s. Most of what remains is fixed — narrowing two multi-million-row 

6389#: frames costs ~0.8 s whatever the sample size — which is why 500 trials only 

6390#: reaches 0.84 s and is not worth the worse sample. It is a screen, not a 

6391#: census: a Trial ID that under-specifies does so systematically, across every 

6392#: reading it merges, so a fair sample of this size finds it. 

6393TRIAL_IDENTITY_SAMPLE = 2000 

6394 

6395 

6396def diagnose_trial_identity( 

6397 words: pd.DataFrame, 

6398 fixations: pd.DataFrame, 

6399 *, 

6400 sample_trials: int | None = None, 

6401 seed: int = 0, 

6402) -> dict[str, object]: 

6403 """Report evidence that one ``trial_id`` covers more than one reading (VAL-7). 

6404 

6405 ``sample_trials`` caps how many trials are examined; ``None`` scans them all 

6406 (what the 🗂️ Data page's *Check every trial* button asks for). The sample is 

6407 drawn with a fixed ``seed``, because a screen whose verdict flickers between 

6408 reruns is worse than no screen. ``sampled_from`` reports the corpus size the 

6409 sample was drawn from, or ``None`` when nothing was sampled — every count in 

6410 the report is then "out of ``trials``", which is what was actually looked at. 

6411 

6412 Returns a dict with: 

6413 

6414 - ``duplicate_word_rows`` — rows sharing ``(key…, word_id)`` in the words 

6415 table. The only *structural* signal: a word box is a property of the 

6416 stimulus, so one row per word per reading is an invariant, not a 

6417 heuristic — and it needs no clock. 

6418 - ``repeated_fixation_id_trials`` / ``backwards_clock_trials`` — the 

6419 independent cross-check, and the one that says *what* merged: a clock 

6420 jumping backwards mid-trial is a second recording starting, not a 

6421 regression. 

6422 - ``multi_valued_columns`` — ``column → number of trials in which it takes 

6423 more than one value``, over :data:`_TRIAL_WITNESS_COLUMNS`. The most 

6424 diagnostic, because the column name is the remedy. 

6425 - ``trials`` / ``affected_trials`` — the denominator, and how many readings 

6426 any signal implicates. 

6427 

6428 Pure and read-only: this reports, it never repairs. Empty frames, or frames 

6429 with no id columns, produce an all-clear rather than an error. 

6430 """ 

6431 report: dict[str, object] = { 

6432 "trials": 0, 

6433 "affected_trials": 0, 

6434 "duplicate_word_rows": 0, 

6435 "repeated_fixation_id_trials": 0, 

6436 "backwards_clock_trials": 0, 

6437 "multi_valued_columns": {}, 

6438 "mixed_source_shapes": False, 

6439 "sampled_from": None, 

6440 } 

6441 key_frame = fixations if fixations is not None and not fixations.empty else words 

6442 if key_frame is None or key_frame.empty: 

6443 return report 

6444 key = trial_identity_key(key_frame) 

6445 if len(key) < 2: # need at least participant + trial to speak of a reading 

6446 return report 

6447 words, fixations, sampled_from = _sample_trials( 

6448 words, fixations, sample_trials, seed 

6449 ) 

6450 report["sampled_from"] = sampled_from 

6451 # The denominator is the union across both frames, not one of them: a corpus 

6452 # can carry word boxes for readers who have no fixations (the bundled demo 

6453 # does), and counting the numerator over words against a fixations-only 

6454 # denominator reads as "6 of 4 trials". 

6455 all_keys: set = set() 

6456 for frame in (words, fixations): 

6457 if frame is None or frame.empty: 

6458 continue 

6459 fkey = trial_identity_key(frame) 

6460 if len(fkey) < 2: 

6461 continue 

6462 sub = frame[fkey].drop_duplicates() 

6463 all_keys |= {tuple(str(v) for v in row) for row in sub.to_numpy()} 

6464 report["trials"] = len(all_keys) 

6465 flagged: set = set() 

6466 

6467 def _keys_of(frame: pd.DataFrame, mask) -> set: 

6468 sub = frame.loc[mask, key].drop_duplicates() 

6469 return {tuple(str(v) for v in row) for row in sub.to_numpy()} 

6470 

6471 # (1) Structural: one word row per word per reading. 

6472 if words is not None and not words.empty and "word_id" in words.columns: 

6473 wkey = trial_identity_key(words) 

6474 if len(wkey) >= 2: 

6475 dup = words.duplicated(subset=wkey + ["word_id"], keep=False) 

6476 report["duplicate_word_rows"] = int(dup.sum()) 

6477 flagged |= _keys_of(words, dup) 

6478 

6479 if fixations is not None and not fixations.empty: 

6480 fkey = trial_identity_key(fixations) 

6481 # (2) A fixation id repeating inside one reading. 

6482 if len(fkey) >= 2 and "fixation_id" in fixations.columns: 

6483 counts = fixations.groupby(fkey, dropna=False)["fixation_id"].agg( 

6484 lambda s: int(s.size - s.nunique()) 

6485 ) 

6486 bad = counts[counts > 0] 

6487 report["repeated_fixation_id_trials"] = len(bad) 

6488 flagged |= {tuple(str(v) for v in _as_tuple(k)) for k in bad.index} 

6489 # (3) A clock that runs backwards mid-reading. 

6490 if len(fkey) >= 2 and "timestamp_ms" in fixations.columns: 

6491 clock = pd.to_numeric(fixations["timestamp_ms"], errors="coerce") 

6492 ordered = pd.DataFrame({"_t": clock.to_numpy()}) 

6493 for col in fkey: 

6494 ordered[col] = fixations[col].astype(str).to_numpy() 

6495 drops = ordered.groupby(fkey, dropna=False)["_t"].apply( 

6496 lambda s: bool((s.diff() < 0).any()) 

6497 ) 

6498 bad_clock = drops[drops] 

6499 report["backwards_clock_trials"] = len(bad_clock) 

6500 flagged |= {tuple(str(v) for v in _as_tuple(k)) for k in bad_clock.index} 

6501 

6502 # (4) A column that should be constant taking several values. 

6503 multi: dict[str, int] = {} 

6504 for frame in (words, fixations): 

6505 if frame is None or frame.empty: 

6506 continue 

6507 fkey = trial_identity_key(frame) 

6508 if len(fkey) < 2: 

6509 continue 

6510 present = [c for c in _TRIAL_WITNESS_COLUMNS if c in frame.columns] 

6511 if not present: 

6512 continue 

6513 nunique = frame.groupby(fkey, dropna=False)[present].nunique(dropna=True) 

6514 for col in present: 

6515 offenders = nunique[col] > 1 

6516 count = int(offenders.sum()) 

6517 if count: 

6518 multi[col] = max(multi.get(col, 0), count) 

6519 flagged |= { 

6520 tuple(str(v) for v in _as_tuple(k)) 

6521 for k in nunique.index[offenders.to_numpy()] 

6522 } 

6523 report["multi_valued_columns"] = dict( 

6524 sorted(multi.items(), key=lambda kv: (-kv[1], kv[0])) 

6525 ) 

6526 # #374 F3: when `source_file` is what varies, the files may be two kinds of 

6527 # table joined into one (a fixation report and an interest-area report) — 

6528 # adding `source_file` to the Trial ID would only hide that. 

6529 report["mixed_source_shapes"] = SOURCE_FILE_COLUMN in multi and any( 

6530 _mixed_source_shapes(frame) for frame in (words, fixations) 

6531 ) 

6532 report["affected_trials"] = len(flagged) 

6533 return report 

6534 

6535 

6536def _mixed_source_shapes(frame: pd.DataFrame | None) -> bool: 

6537 """Whether the files behind ``frame`` filled different columns (#374 F3). 

6538 

6539 Each ``source_file``'s rows are reduced to the set of columns they hold any 

6540 value in; two files of one export share that set, while a fixation report 

6541 and an interest-area report joined into one table do not. 

6542 """ 

6543 if frame is None or frame.empty or SOURCE_FILE_COLUMN not in frame.columns: 

6544 return False 

6545 filled = frame.notna().groupby(frame[SOURCE_FILE_COLUMN], sort=False).any() 

6546 return len(filled.drop_duplicates()) > 1 

6547 

6548 

6549def _as_tuple(key) -> tuple: 

6550 """A groupby index label as a tuple, whether or not it was a MultiIndex.""" 

6551 return key if isinstance(key, tuple) else (key,) 

6552 

6553 

6554def trial_identity_warning(report: dict[str, object]) -> str | None: 

6555 """One line naming the problem and the remedy, or ``None`` when all clear. 

6556 

6557 The column name is the fix, so it leads whenever there is one — "add it to 

6558 the Trial ID mapping" is something the user can act on, where "N trials look 

6559 wrong" is not. 

6560 """ 

6561 if not report or not report.get("affected_trials"): 

6562 return None 

6563 affected = int(report["affected_trials"]) 

6564 total = int(report.get("trials") or 0) 

6565 sampled_from = report.get("sampled_from") 

6566 scope = ( 

6567 f"of {total:,} sampled trials (out of {int(sampled_from):,})" 

6568 if sampled_from 

6569 else f"of {total:,} trials" 

6570 ) 

6571 lead = ( 

6572 f"**{affected:,} {scope} look like more than one trial " 

6573 f"under the current Trial ID.**" 

6574 ) 

6575 multi = report.get("multi_valued_columns") or {} 

6576 if report.get("mixed_source_shapes"): 

6577 return ( 

6578 f"{lead} Their rows come from files with different columns — one " 

6579 "table probably holds both fixation and interest-area reports. " 

6580 "Upload each report in its own row: Fixations, or Words (interest " 

6581 "areas)." 

6582 ) 

6583 if multi: 

6584 col, count = next(iter(multi.items())) 

6585 return ( 

6586 f"{lead} `{col}` takes more than one value inside a trial " 

6587 f"({plural(count, 'trial')}) — adding it to the Trial ID mapping " 

6588 "would separate them." 

6589 ) 

6590 parts = [] 

6591 if report.get("duplicate_word_rows"): 

6592 parts.append(plural(report["duplicate_word_rows"], "duplicated word row")) 

6593 if report.get("repeated_fixation_id_trials"): 

6594 parts.append( 

6595 f"{plural(report['repeated_fixation_id_trials'], 'trial')} with a " 

6596 "repeated fixation ID" 

6597 ) 

6598 if report.get("backwards_clock_trials"): 

6599 parts.append( 

6600 f"{plural(report['backwards_clock_trials'], 'trial')} whose fixation " 

6601 "onsets run backwards" 

6602 ) 

6603 return f"{lead} Evidence: {', '.join(parts)}." 

6604 

6605 

6606def count_trials(words: pd.DataFrame, fixations: pd.DataFrame) -> int: 

6607 """How many distinct ``(participant_id, trial_id)`` trials the frames hold. 

6608 

6609 Counts across both frames, so a words-only or fixations-only dataset is 

6610 measured just as well as a paired one. 

6611 """ 

6612 keys: set = set() 

6613 for df in (words, fixations): 

6614 if df is None or df.empty: 

6615 continue 

6616 if "participant_id" not in df.columns or "trial_id" not in df.columns: 

6617 continue 

6618 keys.update(zip(df["participant_id"].astype(str), df["trial_id"].astype(str))) 

6619 return len(keys) 

6620 

6621 

6622def diagnose_filters( 

6623 words: pd.DataFrame, 

6624 fixations: pd.DataFrame, 

6625 steps: Sequence[tuple], 

6626) -> list[dict]: 

6627 """Attribute an empty trial pool to the filter(s) that caused it (UX-7). 

6628 

6629 ``steps`` is ``(label, apply)`` or ``(label, apply, keys)``, where 

6630 ``apply(words, fixations)`` returns the frames with *only that one* filter 

6631 applied and ``keys`` is the session-state key(s) that filter is stored under 

6632 (so the caller can offer "clear just this one"). Each step is measured against 

6633 the **unfiltered** frames, so the result says what each filter does on its 

6634 own — which is the question a user staring at an empty plot is asking. A step 

6635 that alone leaves nothing is the culprit; if every step leaves something but 

6636 the combination doesn't, it's their intersection, and the caller can say so. 

6637 

6638 Returns one dict per step: ``{"label", "kept", "dropped", "empties", "keys"}``. 

6639 """ 

6640 total = count_trials(words, fixations) 

6641 report: list[dict] = [] 

6642 for step in steps: 

6643 label, apply = step[0], step[1] 

6644 keys = tuple(step[2]) if len(step) > 2 else () 

6645 w, f = apply(words, fixations) 

6646 kept = count_trials(w, f) 

6647 report.append( 

6648 { 

6649 "label": label, 

6650 "kept": kept, 

6651 "dropped": total - kept, 

6652 "empties": total > 0 and kept == 0, 

6653 "keys": keys, 

6654 } 

6655 ) 

6656 return report 

6657 

6658 

6659def filter_raw_gaze( 

6660 raw_gaze: pd.DataFrame, 

6661 participants: list, 

6662 trials: list, 

6663) -> pd.DataFrame: 

6664 """Filter raw gaze data by participants and trials.""" 

6665 if raw_gaze.empty: 

6666 return raw_gaze 

6667 mask = raw_gaze["participant_id"].isin(participants) & raw_gaze["trial_id"].isin( 

6668 trials 

6669 ) 

6670 return raw_gaze[mask] 

6671 

6672 

6673def compute_canvas_size( 

6674 words: pd.DataFrame, fixations: pd.DataFrame 

6675) -> tuple[int, int]: 

6676 """Estimate canvas size from word boxes and fixation extents. 

6677 

6678 Returns the smallest power-of-100 dimensions that comfortably enclose the 

6679 rightmost/bottommost data point. Falls back to DEFAULT_FIGURE_SIZE when 

6680 nothing is available. 

6681 """ 

6682 default_w, default_h = DEFAULT_FIGURE_SIZE 

6683 x_candidates: list[float] = [] 

6684 y_candidates: list[float] = [] 

6685 

6686 def extent(frame: pd.DataFrame, position: str, size: str | None = None) -> float: 

6687 # Coerced (BUG-54): the wizard estimates from the *raw* upload, where a 

6688 # column that merely happens to be named `x` can be text — a 

6689 # decimal-comma export, a unit suffix — and `float(max())` raised. 

6690 if position not in frame.columns: 

6691 return np.nan 

6692 value = _to_number(frame[position]) 

6693 if size is not None and size in frame.columns: 

6694 value = value + _to_number(frame[size]) 

6695 value = value.astype(float) 

6696 return float(value[np.isfinite(value)].max()) 

6697 

6698 if words is not None and not words.empty and "x" in words.columns: 

6699 x_candidates.append(extent(words, "x", "width")) 

6700 y_candidates.append(extent(words, "y", "height")) 

6701 if fixations is not None and not fixations.empty and "x" in fixations.columns: 

6702 x_candidates.append(extent(fixations, "x")) 

6703 y_candidates.append(extent(fixations, "y")) 

6704 # NaN maxima happen when fixations ship without coordinates (AOI-sequence 

6705 # data) and no word boxes were available to fill them in. 

6706 x_candidates = [v for v in x_candidates if np.isfinite(v)] 

6707 y_candidates = [v for v in y_candidates if np.isfinite(v)] 

6708 if not x_candidates or not y_candidates: 

6709 return max(int(default_w), 100), max(int(default_h), 100) 

6710 width = int(np.ceil(max(x_candidates) / 100.0) * 100) 

6711 height = int(np.ceil(max(y_candidates) / 100.0) * 100) 

6712 return max(width, 100), max(height, 100) 

6713 

6714 

6715def canvas_geometry_frames( 

6716 words: pd.DataFrame | None, 

6717 word_schema: dict | None, 

6718 fixations: pd.DataFrame | None, 

6719 fixation_schema: dict | None, 

6720) -> tuple[pd.DataFrame, pd.DataFrame]: 

6721 """The mapped geometry of *raw* tables, in the canonical columns 

6722 :func:`compute_canvas_size` reads (DATA-46). 

6723 

6724 The add-dataset wizard asks for the screen before anything is normalized, so 

6725 it only has the upload as read — ``IA_LEFT`` and ``CURRENT_FIX_X``, not 

6726 ``x``. Handed straight to :func:`compute_canvas_size`, an EyeLink export has 

6727 no column called ``x``, and the "estimate" was the default screen under 

6728 another name. This projects just the mapped coordinate columns (word boxes 

6729 as edges *or* origin + size, fixation x/y) onto ``x``/``y``/``width``/ 

6730 ``height`` — cheap, and correct for any mapping the user has picked so far. 

6731 A field that is not mapped yet is simply absent. 

6732 """ 

6733 

6734 def column(frame: pd.DataFrame, schema: dict, key: str): 

6735 name = schema.get(key) 

6736 if not isinstance(name, str) or name not in frame.columns: 

6737 return None 

6738 return _to_number(frame[name]) 

6739 

6740 word_geometry = pd.DataFrame() 

6741 if words is not None and not words.empty and word_schema: 

6742 left, right = ( 

6743 column(words, word_schema, "left"), 

6744 column(words, word_schema, "right"), 

6745 ) 

6746 top, bottom = ( 

6747 column(words, word_schema, "top"), 

6748 column(words, word_schema, "bottom"), 

6749 ) 

6750 if left is not None and right is not None: 

6751 word_geometry["x"], word_geometry["width"] = left, right - left 

6752 elif (x := column(words, word_schema, "x")) is not None: 

6753 word_geometry["x"] = x 

6754 if (width := column(words, word_schema, "width")) is not None: 

6755 word_geometry["width"] = width 

6756 if top is not None and bottom is not None: 

6757 word_geometry["y"], word_geometry["height"] = top, bottom - top 

6758 elif (y := column(words, word_schema, "y")) is not None: 

6759 word_geometry["y"] = y 

6760 if (height := column(words, word_schema, "height")) is not None: 

6761 word_geometry["height"] = height 

6762 

6763 fixation_geometry = pd.DataFrame() 

6764 if fixations is not None and not fixations.empty and fixation_schema: 

6765 x, y = ( 

6766 column(fixations, fixation_schema, "x"), 

6767 column(fixations, fixation_schema, "y"), 

6768 ) 

6769 if x is not None and y is not None: 

6770 fixation_geometry["x"], fixation_geometry["y"] = x, y 

6771 return word_geometry, fixation_geometry 

6772 

6773 

6774# Primary EyeLink IA measures. When a words frame already carries all of these 

6775# (a pre-aggregated export, e.g. OneStop), the fixation-based recompute is a 

6776# fallback whose output is discarded by the "existing values win" merge — so we 

6777# skip it entirely. See compute_per_word_measures for the precedence rule. 

6778_PREAGGREGATED_METRIC_COLUMNS = [ 

6779 "first_fixation_ms", 

6780 "first_pass_gaze_duration_ms", 

6781 "total_fixation_duration_ms", 

6782 "n_fixations", 

6783] 

6784 

6785 

6786def compute_word_metrics(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame: 

6787 """Return per-word reading measures. 

6788 

6789 If the words table already carries pre-aggregated measures (EyeLink IA 

6790 export), those values are preserved. Anything missing is computed from 

6791 fixations + bounding boxes via `measures.compute_per_word_measures`. 

6792 

6793 Cached on a cheap content *fingerprint* of the inputs (see 

6794 ``frame_fingerprint``) rather than a full DataFrame hash, so a rerun that 

6795 doesn't change the data reuses the result without re-hashing millions of 

6796 rows. The frames themselves are passed un-hashed (underscore args). 

6797 """ 

6798 return _compute_word_metrics_cached( 

6799 words, 

6800 fixations, 

6801 cache_key=(frame_fingerprint(words), frame_fingerprint(fixations)), 

6802 ) 

6803 

6804 

6805def preprocess_fixation_stage( 

6806 words: pd.DataFrame, fixations: pd.DataFrame, settings: dict 

6807) -> tuple[pd.DataFrame, pd.DataFrame]: 

6808 """Cached PRE-1 stage; disabled returns the original fixation object.""" 

6809 if not settings.get("enabled"): 

6810 return fixations, pd.DataFrame() 

6811 key = ( 

6812 frame_fingerprint(words), 

6813 frame_fingerprint(fixations), 

6814 tuple(sorted(settings.items())), 

6815 ) 

6816 result = _preprocess_fixation_stage_cached(words, fixations, settings, key) 

6817 # BUG-103: a fresh copy out of the cache each rerun, named by its inputs. 

6818 assign_derived(result, "preprocess_fixation_stage", (words, fixations), settings) 

6819 return result 

6820 

6821 

6822#: Ceiling on the caches keyed by a *single trial's* frames (PERF-6). Without 

6823#: one, a bulk export over OneStop's 20,000 trials leaves 20,000 result frames 

6824#: behind it — each computed once and never asked for again. Large enough that 

6825#: stepping back and forth through a few dozen trials still hits the cache, 

6826#: small enough that an export's footprint is the corpus, not the corpus plus a 

6827#: copy of every trial's measures. 

6828PER_TRIAL_CACHE_ENTRIES = 128 

6829 

6830 

6831@st.cache_data( 

6832 show_spinner="Preprocessing fixations…", max_entries=PER_TRIAL_CACHE_ENTRIES 

6833) 

6834def _preprocess_fixation_stage_cached( 

6835 _words: pd.DataFrame, _fixations: pd.DataFrame, settings: dict, cache_key 

6836) -> tuple[pd.DataFrame, pd.DataFrame]: 

6837 from .measures import assign_fixations_to_words, enrich_fixations 

6838 from .preprocessing import preprocess_fixations 

6839 

6840 assigned = enrich_fixations(assign_fixations_to_words(_fixations, _words), _words) 

6841 return preprocess_fixations(assigned, _words, settings=settings) 

6842 

6843 

6844@st.cache_data( 

6845 show_spinner="Computing reading measures…", max_entries=PER_TRIAL_CACHE_ENTRIES 

6846) 

6847def _compute_word_metrics_cached( 

6848 _words: pd.DataFrame, _fixations: pd.DataFrame, cache_key 

6849) -> pd.DataFrame: 

6850 from .measures import compute_per_word_measures 

6851 

6852 if _words.empty: 

6853 return _words.copy() 

6854 

6855 # Existing IA measures still win column-by-column inside the measure 

6856 # function, but PRE-4 adds measures EyeLink exports do not usually carry. 

6857 # Compute whenever fixations exist so those missing fields are not silently 

6858 # absent merely because the four legacy headline columns were pre-aggregated. 

6859 enriched = ( 

6860 compute_per_word_measures(_fixations, _words) 

6861 if not _fixations.empty 

6862 else _words 

6863 ) 

6864 

6865 metric_fields = [ 

6866 "first_fixation_ms", 

6867 "first_pass_gaze_duration_ms", 

6868 "regression_path_duration_ms", 

6869 "total_fixation_duration_ms", 

6870 "higher_pass_fixation_duration_ms", 

6871 "last_run_dwell_time_ms", 

6872 "n_fixations", 

6873 "skip_flag", 

6874 "regression_in_count", 

6875 "regression_out_count", 

6876 "regression_in_flag", 

6877 "regression_out_flag", 

6878 "trial_dwell_time_ms", 

6879 "trial_fixation_count", 

6880 "trial_ia_count", 

6881 "word_length", 

6882 "word_length_no_punctuation", 

6883 "gaze_duration_ms", 

6884 "initial_landing_position", 

6885 "initial_landing_distance", 

6886 "number_of_regressions_in", 

6887 "second_pass_duration_ms", 

6888 "single_fixation_duration_ms", 

6889 "first_fix_x", 

6890 "first_fix_y", 

6891 "gpt2_surprisal", 

6892 "wordfreq_frequency", 

6893 "subtlex_frequency", 

6894 "universal_pos", 

6895 "ptb_pos", 

6896 "reduced_pos", 

6897 "dependency_relation", 

6898 "morphological_features", 

6899 "entity_type", 

6900 "head_word_index", 

6901 "distance_to_head", 

6902 "left_dependents_count", 

6903 "right_dependents_count", 

6904 ] 

6905 base_fields = [ 

6906 "participant_id", 

6907 "trial_id", 

6908 "text_id", 

6909 "word_id", 

6910 "text", 

6911 "line_idx", 

6912 ] 

6913 present_fields = [ 

6914 col for col in base_fields + metric_fields if col in enriched.columns 

6915 ] 

6916 metrics = enriched[present_fields].copy() 

6917 

6918 numeric_fields = [ 

6919 "first_fixation_ms", 

6920 "first_pass_gaze_duration_ms", 

6921 "regression_path_duration_ms", 

6922 "total_fixation_duration_ms", 

6923 "higher_pass_fixation_duration_ms", 

6924 "last_run_dwell_time_ms", 

6925 "trial_dwell_time_ms", 

6926 "trial_fixation_count", 

6927 "trial_ia_count", 

6928 "regression_in_count", 

6929 "regression_out_count", 

6930 "word_length", 

6931 "word_length_no_punctuation", 

6932 "gaze_duration_ms", 

6933 "first_fix_x", 

6934 "first_fix_y", 

6935 "gpt2_surprisal", 

6936 "wordfreq_frequency", 

6937 "subtlex_frequency", 

6938 "head_word_index", 

6939 "distance_to_head", 

6940 "left_dependents_count", 

6941 "right_dependents_count", 

6942 ] 

6943 for col in numeric_fields: 

6944 if col in metrics.columns: 

6945 metrics[col] = pd.to_numeric(metrics[col], errors="coerce") 

6946 if "n_fixations" in metrics.columns: 

6947 metrics["n_fixations"] = ( 

6948 pd.to_numeric(metrics["n_fixations"], errors="coerce") 

6949 .fillna(0) 

6950 .astype("Int64") 

6951 ) 

6952 for col in ["skip_flag", "regression_in_flag", "regression_out_flag"]: 

6953 if col in metrics.columns: 

6954 # BUG-7: same sentinel-aware coercion as normalization — a frame can 

6955 # reach here carrying raw `'0'` / `'.'` strings (a pre-computed IA 

6956 # measure joined straight in), and a truthiness cast would flag every 

6957 # row True. 

6958 metrics[col] = coerce_flag(metrics[col]) 

6959 return metrics 

6960 

6961 

6962def default_filters(words: pd.DataFrame, fixations: pd.DataFrame) -> dict: 

6963 """Default ("everything selected") filter dict for the current frames. 

6964 

6965 Cached on a cheap content fingerprint so the full-column ``unique()`` scans 

6966 don't re-run on every rerun when the data hasn't changed. 

6967 """ 

6968 return _default_filters_cached( 

6969 words, 

6970 fixations, 

6971 cache_key=(frame_fingerprint(words), frame_fingerprint(fixations)), 

6972 ) 

6973 

6974 

6975@st.cache_data(show_spinner=False) 

6976def _default_filters_cached( 

6977 _words: pd.DataFrame, _fixations: pd.DataFrame, cache_key 

6978) -> dict: 

6979 # UX-166: keyed on the filtered pair, so this misses on every filter change 

6980 # while everything upstream hits — the report shows the gated dataset card 

6981 # while the new pool is worked out, not only once the trial list builds. 

6982 progress.report() 

6983 filters = dict( 

6984 participants=_union_column_values(_words, _fixations, "participant_id"), 

6985 trials=_union_column_values(_words, _fixations, "trial_id"), 

6986 # The participant/trial lists above are the *full* unique set of the 

6987 # (already trial-filtered) frame, so filter_data's membership masks are 

6988 # no-ops — flag that so it can skip the two O(n) scans. 

6989 _participants_cover_all=True, 

6990 _trials_cover_all=True, 

6991 ) 

6992 if "pass_index" in _fixations.columns: 

6993 filters["pass_indices"] = sorted(_fixations["pass_index"].dropna().unique()) 

6994 if "saccade_type" in _fixations.columns: 

6995 filters["saccade_types"] = sorted( 

6996 _fixations["saccade_type"].dropna().astype(str).unique() 

6997 ) 

6998 if "eye" in _fixations.columns: 

6999 filters["eyes"] = sorted(_fixations["eye"].dropna().astype(str).unique()) 

7000 return filters