Coverage for scanpath_studio/data.py: 97%
2596 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1from __future__ import annotations
3import glob
4import hashlib
5import io
6import logging
7import os
8import re
9import string
10import threading
11import uuid
12import warnings
13import weakref
14import zipfile
15from collections import OrderedDict
16from collections.abc import Callable, Hashable, Iterable, Sequence
17from dataclasses import dataclass, field
18from importlib import resources
19from pathlib import Path
20from typing import Any
22import numpy as np
23import pandas as pd
24import streamlit as st
26from . import progress
27from .constants import (
28 DEFAULT_FIGURE_SIZE,
29 PACKAGE_NAME,
30 SAMPLE_INDEX,
31 UPLOAD_FILE_TYPES,
32 plural,
33)
34from .multipart import (
35 CANVAS_HEIGHT,
36 CANVAS_WIDTH,
37 PARENT_KEY,
38 SCREEN_FIXATION_ID,
39 SCREEN_ID,
40 SCREEN_INDEX,
41 SCREEN_TIMESTAMP,
42 grouping_columns,
43 normalize_screen_identity,
44 validate_matching_parts,
45)
47_LOGGER = logging.getLogger(__name__)
49# PERF-3. Per-run memo, `id(frame) -> (frame, fingerprint)`.
50#
51# Once the expensive subtabs went lazy the biggest remaining cost in a rerun was
52# the cache keys themselves: ~26 calls, 43% of the run, and the same handful of
53# frame OBJECTS over and over — every `_c_*` wrapper re-fingerprints the words
54# and fixations frames `app.main` loaded once. Hashing the same object twice in
55# one run cannot produce two answers, so the second hash onwards is pure waste.
56#
57# Entries hold a **weak** reference to the frame, and the `is` re-check below is
58# what makes an `id()` key safe: a collected frame's id may be reissued to
59# another object, but the entry's ref is dead by then, so the lookup misses. A
60# strong reference would work too — and did, at first — but it pins every frame
61# it has seen. That is worse than it sounds here: `st.cache_data` hands out a
62# fresh object per call, so each run's corpus frames are new objects, and a
63# strong memo kept up to `_FINGERPRINT_MEMO_MAX` of them alive from the end of
64# one run until the top of the next — i.e. for a session's whole idle time, on
65# frames that used to be freed the moment `main()` returned.
66#
67# `threading.local` scopes it to the ScriptRunner thread — one per Streamlit
68# session — so two sessions never share entries and one session's reset can't
69# drop another's. `reset_fingerprint_memo()` at the top of `app.main` bounds the
70# staleness window to a single script run. (Widget callbacks run *before* the
71# script body, so the first fingerprints of a rerun — `controls._compute_trial_filters`
72# — still see the previous run's entries. Harmless: an id hit implies object
73# identity either way, and those frames are the ones about to be reused.)
74#
75# THE ASSUMPTION: a fingerprinted frame is not mutated **in place** part-way
76# through a run. That holds today — the frames the app fingerprints are built by
77# `normalize_*` / `filter_*` / `.copy()` and then only read; helpers that add
78# columns (`aggregation.py`) do it to a local copy. A frame with an *assigned*
79# fingerprint (`_STABLE_FINGERPRINTS` below) relies on the same thing for as
80# long as it lives, since its ID is never re-checked against its content.
81# If you ever add an in-place `frame[col] = …` to a fingerprinted frame, either
82# copy instead or the caches downstream of it will serve pre-mutation results.
83_FINGERPRINT_MEMO = threading.local()
84#: Backstop for a non-Streamlit caller (headless `api.py`, the CLI) that never
85#: reaches `reset_fingerprint_memo`: keep the memo from growing without bound.
86#: A rerun makes ~26 calls, so the app never reaches this; a bulk export does
87#: (`compute_word_metrics` adds two entries per trial), which is why eviction is
88#: least-recently-used rather than clear-everything — the latter threw away the
89#: hot corpus frames to make room for per-trial temporaries.
90_FINGERPRINT_MEMO_MAX = 64
93#: PERF-10 → BUG-103: fingerprints the app *knows* rather than computes, so a
94#: large frame is not re-hashed on every rerun. `id → (weakref, fingerprint |
95#: None)`; the weak ref is what makes the id key safe (a reissued id finds a
96#: dead ref), so an entry lives exactly as long as its frame. Three kinds:
97#:
98#: * **assigned** (`assign_fingerprint`): an ID from where the frame came from —
99#: a loader's source token (`stamp_source` / `adopt_source`), a `frame_cache`
100#: slot + key, or a derivation and its parents (`assign_derived`). Free, and
101#: exact as long as the ID determines the content, which each producer
102#: guarantees.
103#: * **vouched** (`vouch_for_frames`, value `None`): a long-lived frame nothing
104#: writes into (tests/test_frame_immutability.py) — hashed in full once, the
105#: first time it is asked for, then remembered.
106#: * anything else is hashed in full, once per run (`_FINGERPRINT_MEMO`).
107#:
108#: An entry is never overwritten: one object keeps one ID, so a frame a
109#: producer hands back unchanged (an input returned as is) keeps its own.
110#: Process-wide on purpose — every kind depends only on content.
111_STABLE_FINGERPRINTS: dict[int, tuple[weakref.ref, tuple | None]] = {}
112#: Dead entries are swept once the registry grows past this. Not a cap — an
113#: entry is only bookkeeping for a frame that is still alive.
114_STABLE_FINGERPRINTS_SWEEP = 64
115#: UX-166 fix-round-2 (Ruling T5-6): guards every iteration/mutation of
116#: `_STABLE_FINGERPRINTS` above. The dict is process-wide, so two script runs
117#: can reach `_register_fingerprints` at once — a superseded run's build publishing
118#: beside the run that replaced it, or two sessions' runs — though no build is
119#: ever shared *across* sessions (`frame_cache`'s identity includes the
120#: session's own store). An unlocked `.items()` iteration racing another
121#: thread's insert raised `RuntimeError: dictionary changed size during
122#: iteration`. Plain `.get` reads (`frame_fingerprint` below) need no lock under
123#: the GIL — only the sweep-and-insert and the write-back do.
124_STABLE_FINGERPRINTS_LOCK = threading.Lock()
127def _frame_parts(value) -> list[tuple[object, pd.DataFrame]]:
128 """``(label, frame)`` for each non-empty frame in a frame, dict or tuple."""
129 if isinstance(value, pd.DataFrame):
130 items = [(0, value)]
131 elif isinstance(value, dict):
132 items = list(value.items())
133 elif isinstance(value, (tuple, list)):
134 items = list(enumerate(value))
135 else:
136 return []
137 return [
138 (label, frame)
139 for label, frame in items
140 if isinstance(frame, pd.DataFrame) and not frame.empty
141 ]
144def _register_fingerprints(entries: list[tuple[pd.DataFrame, tuple | None]]) -> None:
145 """Record known fingerprints, never replacing one a frame already has."""
146 with _STABLE_FINGERPRINTS_LOCK:
147 if len(_STABLE_FINGERPRINTS) > _STABLE_FINGERPRINTS_SWEEP:
148 for dead in [
149 k for k, (ref, _) in _STABLE_FINGERPRINTS.items() if ref() is None
150 ]:
151 _STABLE_FINGERPRINTS.pop(dead, None)
152 for frame, value in entries:
153 current = _STABLE_FINGERPRINTS.get(id(frame))
154 if current is not None and current[0]() is frame:
155 continue
156 _STABLE_FINGERPRINTS[id(frame)] = (weakref.ref(frame), value)
159def vouch_for_frames(value) -> None:
160 """Mark long-lived frames as never written in place (PERF-10).
162 Each is hashed in full the first time its fingerprint is asked for, and that
163 answer is kept for as long as the frame lives — for a frame the app holds
164 run after run without knowing where it came from, such as a stored dataset
165 read back from the recovery cache.
166 """
167 _register_fingerprints([(frame, None) for _, frame in _frame_parts(value)])
170def assign_fingerprint(frame: pd.DataFrame, ident: Hashable) -> None:
171 """Give ``frame`` the fingerprint ``("assigned", ident)`` instead of a hash.
173 BUG-103. ``ident`` must determine the frame's content — two frames with the
174 same ``ident`` are taken to be identical, which is exactly what makes the
175 caches downstream reuse a result. A frame that already has a fingerprint
176 keeps it.
177 """
178 if isinstance(frame, pd.DataFrame) and not frame.empty:
179 _register_fingerprints([(frame, ("assigned", ident))])
182#: BUG-103: the `DataFrame.attrs` entry a cached loader labels its frames with.
183#: Only a carrier: `st.cache_data` hands out a fresh copy per call, and `attrs`
184#: survive the copy, so the label reaches the caller — where `adopt_source`
185#: takes it off again before the frame goes anywhere else. Never read anywhere
186#: but there: pandas copies `attrs` onto every frame derived from this one, so a
187#: label left on would follow a filtered or edited copy that is not the source.
188SOURCE_TOKEN_ATTR = "_sps_source_token"
191def stamp_source(value):
192 """Label a loader's frames with a fresh random token, for `adopt_source`.
194 Call it on the return value **inside** an ``@st.cache_data`` loader: the
195 token is drawn once per real load, so every cache hit carries the same one
196 and a reload — a new upload, another read plan, a cleared cache — a new one.
197 Returns ``value``.
198 """
199 token = uuid.uuid4().hex
200 for label, frame in _frame_parts(value):
201 frame.attrs[SOURCE_TOKEN_ATTR] = (token, label)
202 return value
205def adopt_source(*frames: pd.DataFrame | None) -> None:
206 """Turn a loader's `stamp_source` label into each frame's fingerprint.
208 Call it where a loader's frames come out, before anything derives from them.
209 The label comes off the frame, so nothing made from it inherits the token;
210 the fingerprint stays with this object only.
211 """
212 for frame in frames:
213 if not isinstance(frame, pd.DataFrame):
214 continue
215 token = frame.attrs.pop(SOURCE_TOKEN_ATTR, None)
216 if token is not None:
217 assign_fingerprint(frame, ("source", token))
220def assign_derived(outputs, op: str, parents, params=None) -> None:
221 """Fingerprint ``outputs`` by how they were made, not by hashing them.
223 For a pure step — ``outputs`` determined by ``op``, the ``parents`` frames'
224 content and ``params`` — the ID is ``(op, parent fingerprints, params)``,
225 labelled by each output's position. An output that is one of its parents
226 (a step with nothing to do) keeps the parent's fingerprint. ``params`` must
227 be hashable after `hashable_key`; when it is not, the outputs are simply
228 hashed like any other frame.
229 """
230 parent_ids = {id(p) for p in _as_frames(parents)}
231 if all(id(frame) in parent_ids for _, frame in _frame_parts(outputs)):
232 return # nothing new to name (a step with nothing to do)
233 key = hashable_key(params)
234 if not _plain(key):
235 return
236 parent_keys = tuple(frame_fingerprint(p) for p in _as_frames(parents))
237 # Digested: every cache downstream hashes its key on every call, and the
238 # settings can be long (the filter's full trial list), so the ID stays
239 # small however much went into it. `repr` is stable for what `hashable_key`
240 # leaves: tuples of plain values, sets and dicts already sorted.
241 digest = hashlib.blake2b(
242 repr((op, parent_keys, key)).encode(), digest_size=16
243 ).hexdigest()
244 for label, frame in _frame_parts(outputs):
245 if id(frame) not in parent_ids:
246 assign_fingerprint(frame, ("derived", op, digest, label))
249#: What a derived ID's settings may hold: values whose `repr` is their content.
250#: Anything else — an object whose `repr` is its address, a Series — could
251#: name two different settings the same way, so it is hashed instead.
252_PLAIN_TYPES = (str, int, float, bool, type(None), np.generic, pd.Timestamp)
255def _plain(value) -> bool:
256 if isinstance(value, tuple):
257 return all(_plain(v) for v in value)
258 return isinstance(value, _PLAIN_TYPES)
261def _as_frames(value) -> tuple:
262 if isinstance(value, pd.DataFrame) or value is None:
263 return (value,)
264 return tuple(value)
267def hashable_key(value):
268 """``value`` as a hashable, order-free key part (sets and dicts sorted)."""
269 if isinstance(value, dict):
270 return tuple(sorted(((str(k), hashable_key(v)) for k, v in value.items())))
271 if isinstance(value, (set, frozenset)):
272 return tuple(sorted((hashable_key(v) for v in value), key=repr))
273 if isinstance(value, (list, tuple)):
274 return tuple(hashable_key(v) for v in value)
275 return value
278#: Session-state home of the no-copy frame caches (PERF-6), one entry per slot.
279_FRAME_CACHE_KEY = "_sps_frame_cache"
281#: UX-166 "latest request wins" (T5-1): the store key `(_LATEST_REQUESTED,
282#: slot)` — a tuple, so it can never collide with a real (string) slot name —
283#: holds the most recently *requested* key for that slot, recorded by
284#: `frame_cache` on every call, hit or miss. A build that finishes only writes
285#: `store[slot]` while this still names its own key; otherwise a newer request
286#: has already been *made* — whether or not it has itself finished yet, or
287#: ever will — and this build's (still-valid, still returned to its own
288#: caller) result must not clobber it.
289_LATEST_REQUESTED = "__requested__"
291#: PERF-18: the store key `(_EARLIER, slot)` holds a slot's *earlier* entries
292#: — `(key, value)` pairs, most recent first — when it was asked to `keep`
293#: more than one. `store[slot]` stays the current entry, so a one-entry slot
294#: looks exactly as it always did.
295_EARLIER = "__earlier__"
298@dataclass
299class _InFlight:
300 done: threading.Event = field(default_factory=threading.Event)
301 value: Any = None
302 ok: bool = False
303 #: Set only when the owner's build raised an ordinary ``Exception`` — never
304 #: for a ``BaseException`` that isn't one (`progress.Cancelled`, Streamlit's
305 #: `StopException`). See `_shared_build`.
306 error: Exception | None = None
309#: UX-166: builds in progress, so a rerun that asks for the same frame waits
310#: for the one already running instead of starting a second. A click during a
311#: long load abandons the running script and starts a new one at once
312#: (`runner.fastReruns`); without this the new run normalized the corpus again
313#: beside the first. `st.cache_data` has the same guarantee through its own
314#: per-key lock.
315_INFLIGHT: dict[tuple, _InFlight] = {}
316_INFLIGHT_LOCK = threading.Lock()
318#: `lookup`/`publish` (see `_shared_build`) return/accept this to mean "no
319#: cached value" — never `None`, since a legitimate result can itself be `None`.
320_MISSING = object()
323def _shared_build(
324 ident: tuple,
325 build: Callable[[], Any],
326 *,
327 lookup: Callable[[], Any] | None = None,
328 publish: Callable[[Any], None] | None = None,
329) -> Any:
330 """``build()``, run once for everyone asking for ``ident`` at the same time.
332 A caller that finds a build running waits for it and reuses its result.
333 If the owner's build raises an ordinary ``Exception``, that same exception
334 is re-raised in every waiter too — the input hasn't changed, so rebuilding
335 would just fail again the same way. Only a ``BaseException`` that is *not*
336 an ``Exception`` (`progress.Cancelled`, Streamlit's `StopException`) means
337 nobody actually finished the build, so a waiter then builds it itself.
339 ``lookup``/``publish`` let a cache-shaped caller close UX-166's "latest
340 request wins" race: a new owner calls ``lookup()`` right after winning
341 ownership — a value another, faster build already published for this
342 exact ``ident`` a moment earlier is reused without rebuilding — and a
343 successful build calls ``publish(value)`` *before* the in-flight entry is
344 popped, so "is this result still wanted, or has a newer request for this
345 slot already been made" is decided while this ``ident`` still has exactly
346 one owner. A joined waiter calls its own ``publish(entry.value)`` too
347 (UX-166 fix-round-2, Minor #1 of Ruling T5-5): the owner's own decision was
348 made against whatever key was latest *then*, and a request for this exact
349 ``ident`` can itself become the latest again before the owner's entry is
350 popped — without this, that waiter would still get the right *value* back
351 but the store would never hold it.
353 Neither ``lookup`` nor ``publish`` is called for a plain (non-cache) use
354 of this function, and neither's own failure is allowed to leak the
355 in-flight entry or hang every waiter forever (UX-166 fix-round-2, Ruling
356 T5-5 — a24e105 always popped the entry and signalled ``done``; the "latest
357 request wins" fix lost that guarantee by writing the cleanup out per path
358 instead of in one ``finally``): a failing ``lookup`` is treated as a miss
359 (logged at debug, then built normally); a failing ``publish`` is logged as
360 a warning and swallowed — the build itself already succeeded, ``entry.ok``
361 is already ``True``, and its caller still gets its value either way.
362 """
363 while True:
364 with _INFLIGHT_LOCK:
365 entry = _INFLIGHT.get(ident)
366 owner = entry is None
367 if owner:
368 entry = _InFlight()
369 _INFLIGHT[ident] = entry
370 if owner:
371 # UX-166 fix-round-2 (Ruling T5-5): the whole owner branch is one
372 # try/finally, so the registry pop and `done.set()` ALWAYS run —
373 # whether `lookup`, `build` or `publish` raises, or nothing does.
374 # Without this, a raising `publish` (the reviewer's repro:
375 # `_register_fingerprints` racing another session's concurrent insert)
376 # left the entry registered forever: every waiter already joined
377 # blocks in `entry.done.wait()` with no timeout and no Streamlit
378 # checkpoint to free it, and every later miss for this `ident`
379 # joins the same dead entry and hangs too.
380 try:
381 hit = _MISSING
382 if lookup is not None:
383 try:
384 hit = lookup()
385 except Exception:
386 _LOGGER.debug(
387 "_shared_build lookup failed for %r; building instead",
388 ident,
389 exc_info=True,
390 )
391 hit = _MISSING
392 if hit is not _MISSING:
393 value = hit
394 else:
395 try:
396 value = build()
397 except BaseException as exc:
398 if isinstance(exc, Exception):
399 entry.error = exc
400 raise
401 entry.value = value
402 entry.ok = True
403 # Only a build we actually ran gets published — a lookup hit
404 # means the store already holds this exact key's value.
405 if hit is _MISSING and publish is not None:
406 try:
407 publish(value)
408 except Exception:
409 # A WARNING, not DEBUG (the in-app log captures from
410 # INFO): swallowed, a publish that keeps failing shows
411 # only as every rerun rebuilding.
412 _LOGGER.warning(
413 "_shared_build publish failed for %r; the built "
414 "value is still returned, just not cached",
415 ident,
416 exc_info=True,
417 )
418 return value
419 finally:
420 with _INFLIGHT_LOCK:
421 _INFLIGHT.pop(ident, None)
422 entry.done.set()
423 # UX-166: waiting on a build another run started is this run's work too
424 # — the owner reports into its own task, so without this a gated card
425 # over the wait (the Corpus measures, opened afresh each run) never
426 # shows. It is also a cancel checkpoint: a waiter whose own task was
427 # cancelled stops here instead of waiting out a build it no longer
428 # wants. A hit returned above, so an all-hit rerun never gets here.
429 progress.report()
430 entry.done.wait()
431 if entry.ok:
432 if publish is not None:
433 try:
434 publish(entry.value)
435 except Exception:
436 _LOGGER.warning(
437 "_shared_build waiter publish failed for %r",
438 ident,
439 exc_info=True,
440 )
441 return entry.value
442 if entry.error is not None:
443 raise entry.error
444 # The owner was cancelled or stopped, not merely wrong: nobody actually
445 # built this. Loop back and become the new owner ourselves.
448def frame_cache(slot: str, key, build, *, keep: int = 1):
449 """Return ``build()``'s result, reusing the last one while ``key`` holds.
451 ``st.cache_data`` hands every caller a private **deep copy** of its result.
452 That is the right default — it stops one part of the app corrupting
453 another's data — but for the corpus-scale frames it buys nothing and costs a
454 great deal: measured on the full OneStop reports, ~1.15 s and ~1.2 GB of
455 allocate-and-discard on *every* rerun, i.e. on every widget touch, for data
456 the app already had.
458 Nothing writes into those frames in place. That is not an assumption:
459 ``tests/test_frame_immutability.py`` captures them as the loader yields them,
460 runs a full render and a rerun on top, and asserts they come back
461 byte-identical — and the same file's canary proves the check would notice if
462 they didn't. So this hands back **the object itself**.
464 One entry per slot by default, deliberately: keeping the previous corpus
465 alive beside the current one costs memory. ``keep`` raises that for a slot
466 where going back is the common move (PERF-18: the normalized pair keeps
467 two, so switching PoTeC → OneStop → PoTeC doesn't normalize PoTeC again —
468 a ~20 s wait at OneStop scale). Falls back to
469 calling ``build`` when there is no session state, which is what the headless
470 API and the CLI see.
472 Keys must be hashable (they are compared with ``==`` and stored as dict
473 keys). Concurrent requests for the same slot + key share one build in
474 flight (`_shared_build`); a build that finishes only *publishes* — writes
475 the entry every later request for that key reuses — while its key is still
476 the slot's most recently requested one (UX-166's "latest request wins"),
477 so a superseded build's late finish can never clobber a newer result. It
478 still returns its value to its own caller either way.
479 """
480 try:
481 store = st.session_state.setdefault(_FRAME_CACHE_KEY, {})
482 except (RuntimeError, AttributeError, KeyError) as exc:
483 # No Streamlit runtime (api.py, cli.py, a bare import) — nothing to
484 # cache into, and nothing that reruns. Logged rather than silent: the
485 # symptom of catching this wrongly is a cache that simply never works,
486 # which is invisible except as everything being slow.
487 _LOGGER.debug("frame_cache falling back to a plain call: %s", exc)
488 return build()
489 # UX-166: record this as the slot's latest request on EVERY call — a hit
490 # included, since "the user cancelled back to an earlier dataset" is a hit
491 # for the slot's *current* entry, and without recording it here too an
492 # abandoned build for a *different* key would still look, to its own late
493 # `publish`, like nobody had asked for anything else since.
494 with _INFLIGHT_LOCK:
495 store[(_LATEST_REQUESTED, slot)] = key
496 entry = store.get(slot)
497 if entry is not None and entry[0] == key:
498 return entry[1]
499 if keep > 1:
500 with _INFLIGHT_LOCK:
501 earlier = store.get((_EARLIER, slot)) or []
502 for index, (old_key, old_value) in enumerate(earlier):
503 if old_key == key:
504 # Promote it back to current; the entry it replaces
505 # becomes the most recent earlier one.
506 rest = earlier[:index] + earlier[index + 1 :]
507 current = store.get(slot)
508 store[(_EARLIER, slot)] = ([current] if current else []) + rest
509 store[slot] = (old_key, old_value)
510 return old_value
512 def _lookup() -> Any:
513 # UX-166: a new owner re-checks the store before building — a
514 # concurrent build for this exact key may have just published,
515 # between our own miss above and winning ownership below.
516 with _INFLIGHT_LOCK:
517 current = store.get(slot)
518 if current is not None and current[0] == key:
519 return current[1]
520 return _MISSING
522 def _publish(value: Any) -> None:
523 with _INFLIGHT_LOCK:
524 wins = store.get((_LATEST_REQUESTED, slot)) == key
525 if wins:
526 previous = store.get(slot)
527 if keep > 1 and previous is not None and previous[0] != key:
528 earlier = [
529 item
530 for item in store.get((_EARLIER, slot)) or []
531 if item[0] != key
532 ]
533 store[(_EARLIER, slot)] = [previous, *earlier][: keep - 1]
534 store[slot] = (key, value)
535 if wins:
536 # BUG-103: the key decides the value, so it is the value's ID too.
537 _register_fingerprints(
538 [
539 (frame, ("assigned", ("frame_cache", slot, key, label)))
540 for label, frame in _frame_parts(value)
541 ]
542 )
544 # UX-166: shared with a build already running for this session, slot and key.
545 return _shared_build(
546 (id(store), slot, key), build, lookup=_lookup, publish=_publish
547 )
550def clear_frame_cache() -> None:
551 """Drop every no-copy frame cache entry (PERF-6).
553 Called by ``app.clear_computation_cache`` alongside ``st.cache_data.clear``:
554 the normalized frames are cached *here* rather than there, so clearing only
555 Streamlit's cache would leave a just-deleted dataset's frames alive.
556 """
557 try:
558 st.session_state.pop(_FRAME_CACHE_KEY, None)
559 except (RuntimeError, AttributeError, KeyError):
560 pass # No runtime: there is no cache to clear.
563def reset_fingerprint_memo() -> None:
564 """Drop the per-run fingerprint memo. Called once per script run (PERF-3)."""
565 getattr(_FINGERPRINT_MEMO, "cache", {}).clear()
568def frame_fingerprint(df: pd.DataFrame | None) -> tuple:
569 """Exact, content-sensitive identity for a DataFrame.
571 Used as an *explicit* ``@st.cache_data`` key for functions that take an
572 underscore-prefixed (un-hashed) frame argument — so Streamlit never re-hashes
573 a multi-million-row frame on every rerun just to look up the cache.
575 **Two frames share a fingerprint only when they are the same data**
576 (BUG-103). It is one of:
578 * the ID the app *assigned* the frame — from the loader that read it, the
579 ``frame_cache`` entry that holds it, or the step that derived it (see
580 ``_STABLE_FINGERPRINTS``). That is what keeps a corpus-sized frame from
581 being hashed on every rerun;
582 * otherwise ``(rows, columns, digest)`` over **every** row. Until BUG-103 a
583 frame over 200,000 rows was keyed on ~384 sampled rows, so a corrected
584 re-upload of the same shape matched the old file and was served its
585 results — and its normalized tables, which were then stored as the new
586 dataset — without a word. A full hash costs ~60 ms per million rows, which
587 is why the large frames on the rerun path are assigned an ID instead.
589 The per-row hashes are digested **in order** rather than summed. Summing is
590 order-invariant, so a frame and a ``sort_values`` of itself — same rows, same
591 index labels, different order — used to share a key while producing different
592 results downstream.
594 ``hash_pandas_object`` raises on columns of unhashable objects (lists/arrays —
595 e.g. parquet-preserved span-index fields). We stringify and retry rather than
596 drop the content signal entirely: zeroing the hash would collapse every frame
597 of the same shape + columns to one fingerprint, serving stale cached results
598 when switching between two such frames. If even that fails the key becomes
599 *unique* rather than zero — an unhashable frame must miss the cache, not
600 match every other frame of its shape.
602 **The same frame object is only hashed once per run** — see the
603 ``_FINGERPRINT_MEMO`` note above for why that is safe and what it assumes.
604 """
605 if df is None or getattr(df, "empty", True):
606 return (0, ())
607 memo = getattr(_FINGERPRINT_MEMO, "cache", None)
608 if memo is None:
609 memo = _FINGERPRINT_MEMO.cache = OrderedDict()
610 key = id(df)
611 hit = memo.get(key)
612 if hit is not None and hit[0]() is df:
613 memo.move_to_end(key)
614 return hit[1]
615 stable = _STABLE_FINGERPRINTS.get(key)
616 if stable is not None and stable[0]() is not df:
617 stable = None
618 if stable is not None and stable[1] is not None:
619 return stable[1]
620 value = _compute_frame_fingerprint(df)
621 if stable is not None:
622 with _STABLE_FINGERPRINTS_LOCK:
623 _STABLE_FINGERPRINTS[key] = (stable[0], value)
624 # Drop entries whose frame has already been collected before evicting a live
625 # one — those are pure bookkeeping and cost nothing to lose.
626 if len(memo) >= _FINGERPRINT_MEMO_MAX:
627 for dead in [k for k, (ref, _) in memo.items() if ref() is None]:
628 del memo[dead]
629 while len(memo) >= _FINGERPRINT_MEMO_MAX:
630 memo.popitem(last=False)
631 memo[key] = (weakref.ref(df), value)
632 return value
635def _compute_frame_fingerprint(df: pd.DataFrame) -> tuple:
636 """The actual hash behind :func:`frame_fingerprint`, memo aside."""
637 cols = tuple(map(str, df.columns))
638 n = len(df)
640 def _hash(frame: pd.DataFrame) -> str:
641 try:
642 per_row = pd.util.hash_pandas_object(frame, index=True)
643 except TypeError:
644 # Unhashable cell objects — stringify so content still drives the key.
645 per_row = pd.util.hash_pandas_object(frame.astype(str), index=True)
646 # Digest the per-row hashes in ORDER. Summing them (the previous
647 # approach) is order-invariant, so a frame and a `sort_values` of itself
648 # — same rows, same index labels, different order — produced the same
649 # key while yielding different results downstream.
650 return hashlib.blake2b(
651 per_row.to_numpy(dtype="uint64").tobytes(), digest_size=16
652 ).hexdigest()
654 try:
655 return (n, cols, _hash(df))
656 except Exception:
657 # Fail CLOSED: a key nothing else can equal, so this frame simply doesn't
658 # share a cache entry. `(n, cols, 0, 0)` failed *open* — every frame of
659 # the same shape collided.
660 return (n, cols, uuid.uuid4().hex)
663# ---------------------------------------------------------------------------
664# Server-side OneStop data source.
665#
666# When the env var `ONESTOP_DATA_DIR` points at a OneStop lacclab export
667# folder (containing `ia_Paragraph.csv.zip` and `fixations_Paragraph.csv.zip`),
668# `load_onestop_server_bundle()` returns them as the (words, fixations) tuple
669# the rest of the pipeline expects. The schema is identical to the bundled
670# sample (the sample is a 3-pid subset of OneStop), so no extra normalisation
671# is required.
672#
673# Drives the "OneStop server bundle" data source option in app.py, used by
674# an external review-app deep-link integration (single pid+trial into this UI).
675# ---------------------------------------------------------------------------
677ONESTOP_DATA_DIR_ENV = "ONESTOP_DATA_DIR"
680def onestop_data_dir() -> Path | None:
681 """Resolved value of `$ONESTOP_DATA_DIR`, or `None` if unset/blank."""
682 raw = os.environ.get(ONESTOP_DATA_DIR_ENV, "").strip()
683 return Path(raw) if raw else None
686def onestop_full_bundle_exists() -> bool:
687 """True when the full OneStop CSV.zip exports are present (not just per-pid
688 shards).
690 When the whole corpus is available the app loads it once and filters in-app
691 (so switching participant is instant); a shards-only setup must instead load
692 one participant's shard at a time (it can't materialize the ~60 GB corpus)."""
693 base = onestop_data_dir()
694 if base is None:
695 return False
696 return (base / "ia_Paragraph.csv.zip").exists() and (
697 base / "fixations_Paragraph.csv.zip"
698 ).exists()
701def _onestop_shard_paths(base: Path, pid: str) -> tuple[Path, Path]:
702 """Resolved per-participant shard paths under `<base>/by_pid/`."""
703 pid = pid.strip().lower()
704 return (
705 base / "by_pid" / "ia" / f"{pid}.parquet",
706 base / "by_pid" / "fixations" / f"{pid}.parquet",
707 )
710def onestop_data_provenance(participant: str | None = None) -> dict:
711 """Where the currently-loaded OneStop data came from, for the Raw Data tab.
713 Parses `ONESTOP_DATA_DIR` (typically `…/onestop_<cohort>/reports/<source>/<date>/full/`)
714 to surface cohort, export source (lacclab / public / osf), and date in the
715 UI so reviewers can verify they're looking at the right export. Also
716 reports the per-pid shard's mtime when a participant is set — that's the
717 timestamp of the actual data the page is currently rendering.
719 Returns an empty dict when `ONESTOP_DATA_DIR` is unset (i.e. the OneStop
720 data source isn't in use — caller should suppress the provenance panel).
721 """
722 base = onestop_data_dir()
723 if base is None:
724 return {}
726 info: dict = {"data_dir": str(base)}
728 # Best-effort parse of the canonical path layout.
729 parts = base.resolve().parts
730 try:
731 # Look for "reports" anchor and grab source/date after it.
732 i = parts.index("reports")
733 info["source"] = parts[i + 1] # lacclab / public / osf
734 info["date"] = parts[i + 2] # YYYYMMDD
735 except (ValueError, IndexError):
736 pass
737 for p in parts:
738 if p.startswith("onestop_"):
739 info["cohort"] = p.removeprefix("onestop_") # L1 / L2
740 break
742 # Reports the per-pid shard's mtime when a participant is set — that's the
743 # timestamp of the bytes the page is rendering right now.
744 if participant:
745 ia_shard, fix_shard = _onestop_shard_paths(base, participant)
746 info["loaded_from"] = "per-pid shard"
747 info["ia_shard"] = str(ia_shard)
748 info["fix_shard"] = str(fix_shard)
749 if ia_shard.is_file():
750 info["ia_shard_mtime"] = ia_shard.stat().st_mtime
751 if fix_shard.is_file():
752 info["fix_shard_mtime"] = fix_shard.stat().st_mtime
753 else:
754 ia_csv = base / "ia_Paragraph.csv.zip"
755 fix_csv = base / "fixations_Paragraph.csv.zip"
756 info["loaded_from"] = "full CSV.zip export"
757 if ia_csv.is_file():
758 info["ia_shard"] = str(ia_csv)
759 info["ia_shard_mtime"] = ia_csv.stat().st_mtime
760 if fix_csv.is_file():
761 info["fix_shard"] = str(fix_csv)
762 info["fix_shard_mtime"] = fix_csv.stat().st_mtime
763 return info
766# UX-166: the dataset card lists this step.
767@st.cache_data(show_spinner=False)
768def load_onestop_server_bundle(
769 participant: str | None = None,
770) -> tuple[pd.DataFrame, pd.DataFrame]:
771 """Load OneStop lacclab IA + fixation reports from `$ONESTOP_DATA_DIR`.
773 Fast path — when `participant` is given and per-pid shards exist under
774 `<ONESTOP_DATA_DIR>/by_pid/{ia,fixations}/<pid>.parquet`, load just that
775 one participant (sub-second). Shards are generated by
776 `python -m scanpath_studio.onestop_shard --data-dir <ONESTOP_DATA_DIR>`.
778 Slow path — fall back to loading the full CSV.zip exports (~3 min, ~60 GB
779 RAM for the L2 cohort). Used when no participant is specified, or when
780 a deep link points at a pid whose shard hasn't been generated yet.
781 """
782 progress.report() # UX-166: a miss — real work, so the gated card may show
783 base = onestop_data_dir()
784 if base is None:
785 return pd.DataFrame(), pd.DataFrame()
787 # Fast path: per-pid shards.
788 if participant:
789 ia_shard, fix_shard = _onestop_shard_paths(base, participant)
790 ia_present = ia_shard.exists()
791 fix_present = fix_shard.exists()
792 if ia_present and fix_present:
793 # PERF-6: parse only the columns the mapping + registry keep.
794 return stamp_source(
795 (
796 read_mapped_table(ia_shard, kind="words"),
797 read_mapped_table(fix_shard, kind="fixations"),
798 )
799 )
800 # NEVER fall through to the 15 GB load when a participant is named —
801 # the deep link is for one pid only, so loading the whole cohort just
802 # to discover the pid still has no data is pure waste. Surface a clear
803 # error and stop. Common cause: pid was excluded from the IA report
804 # (no exported reading data), or shards haven't been generated yet.
805 missing = [
806 f"{p.parent.name}/{p.name}"
807 for p, ok in [(ia_shard, ia_present), (fix_shard, fix_present)]
808 if not ok
809 ]
810 st.error(
811 f"No data for participant {participant!r} on this server: they have "
812 f"no reading data, or its files ({', '.join(missing)}) are missing. "
813 "Whoever runs the server can regenerate them with "
814 "`python -m scanpath_studio.onestop_shard --data-dir <ONESTOP_DATA_DIR>`."
815 )
816 st.stop()
818 # Slow path: full CSV.zip load.
819 ia_path = base / "ia_Paragraph.csv.zip"
820 fix_path = base / "fixations_Paragraph.csv.zip"
821 if not ia_path.exists() or not fix_path.exists():
822 st.error(
823 f"OneStop data not found under {base}. Expected ia_Paragraph.csv.zip + "
824 f"fixations_Paragraph.csv.zip."
825 )
826 return pd.DataFrame(), pd.DataFrame()
827 # PERF-6: an IA report ships 174 columns and a fixation report 300, of
828 # which normalization keeps 27 and 17. Parsing the rest is what made this
829 # path reach ~25 GB resident before a single measure was computed.
830 words = read_mapped_table(ia_path, kind="words")
831 fixations = read_mapped_table(fix_path, kind="fixations")
832 return stamp_source((words, fixations))
835_TRAILING_UNIT = re.compile(r"\s*[\[(][^\[\]()]*[\])]\s*$")
838def _norm_col(name) -> str:
839 """Fold a column name to its case- and separator-insensitive key.
841 Lowercases and drops every non-alphanumeric char, so ``IA_LEFT``,
842 ``ia_left``, ``Ia-Left`` and ``ia left`` all collapse to ``ialeft`` —
843 letting auto-detection match real-world column names that differ only in
844 capitalization or word separators.
846 A trailing unit or coordinate-system block is dropped first (DATA-25):
847 Tobii Pro Lab writes ``Fixation point X [DACS px]``, SMI BeGaze
848 ``Fixation Duration [ms]``, Pupil Labs Neon ``fixation x [px]`` and Tobii
849 Studio ``FixationPointX (MCSpx)``. Folding the brackets' *letters* into the
850 key (``fixationpointxdacspx``) made every one of them miss a candidate that
851 names the same thing, so the export of three major vendors landed in the
852 manual mapping step. Only one trailing block goes — a bracket in the middle
853 of a name is part of the name."""
854 text = _TRAILING_UNIT.sub("", str(name))
855 return re.sub(r"[^a-z0-9]", "", text.lower())
858_COL_SEPARATORS = re.compile(r"[^a-zA-Z0-9]+")
859_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
862def _col_tokens(name) -> list[str]:
863 """Split a raw column name into its separator-delimited tokens (DATA-25
864 second pass), each folded like ``_norm_col``.
866 The trailing-unit block is dropped from the *whole* name first, same as
867 ``_norm_col`` — so a vendor's ``LEFT_px`` tokenizes to ``["left", "px"]``
868 with the unit noise already gone, not left to coincidentally never match
869 a candidate."""
870 text = _TRAILING_UNIT.sub("", str(name))
871 # DATA-57: a CamelCase name (`BoxLeft`, `AoiTop`) has no separator to split
872 # on, so a lower→upper case change counts as one.
873 text = _CAMEL_BOUNDARY.sub(" ", text)
874 return [tok.lower() for tok in _COL_SEPARATORS.split(text) if tok]
877def pick_column(df: pd.DataFrame, candidates: Iterable[str]) -> str | None:
878 """Return the first matching column name from a candidate list.
880 Matching is case- and separator-insensitive (see ``_norm_col``). Candidate
881 order is still priority order — the first candidate with any match wins (so
882 EyeLink names keep beating Gazepoint), and among equally-normalized columns
883 the leftmost one wins.
885 If nothing matches exactly, a second pass (DATA-25) catches a vendor
886 prefix or suffix on a known name — ``AOI_LEFT``, ``LEFT_px`` — by
887 splitting each column on its separators and checking whether any *whole*
888 token equals a candidate. There is no prefix vocabulary to maintain, and
889 no substring matching, so ``top`` never matches ``stop_time`` (the whole
890 token is ``stop``, not ``top``) and ``id`` never matches ``guid``. That
891 still leaves real ambiguity — ``top_left_x`` and ``top_left_y`` both
892 contain the token ``left``; ``max_x`` and ``fix_x`` both contain ``x`` —
893 so the second pass is accepted only when it turns up **exactly one**
894 column across every candidate in the list. Two or more survivors is
895 ambiguity, and ambiguity means the manual mapping step, not a guess: the
896 safety here is uniqueness, not a whitelist.
898 Several survivors get one narrowing step (DATA-60) before that verdict:
899 keep only the ones spelled **entirely** in the list's own words — every
900 token of the column is a token of some candidate. An AOI export's
901 ``AOI_ID`` beside ``AOI_LEFT`` … ``AOI_BOTTOM`` is the case: all five carry
902 the word-id candidate ``aoi``, but only ``AOI_ID``'s other token (``id``,
903 from ``word_id`` / ``IA_ID``) belongs to the word-id list, while ``left``
904 does not. The same uniqueness applies to what is left: one column, or
905 none."""
906 lookup: dict[str, str] = {}
907 for col in df.columns:
908 lookup.setdefault(_norm_col(col), col)
909 candidates = list(candidates)
910 for name in candidates:
911 hit = lookup.get(_norm_col(name))
912 if hit is not None:
913 return hit
915 normed_candidates = {_norm_col(name) for name in candidates}
916 survivors = [col for col in df.columns if normed_candidates & set(_col_tokens(col))]
917 if len(survivors) > 1:
918 vocabulary = {tok for name in candidates for tok in _col_tokens(name)}
919 survivors = [col for col in survivors if set(_col_tokens(col)) <= vocabulary]
920 if len(survivors) == 1:
921 return survivors[0]
922 return None
925#: Milliseconds per unit, for a time column whose header names its unit —
926#: Tobii Pro Lab's `Recording timestamp [μs]`, Pupil Labs Neon's
927#: `start timestamp [ns]` (DATA-40). DATA-25 taught auto-detection to look past
928#: that block; this is what reads it.
929_TIME_UNIT_MS = {
930 "s": 1000.0,
931 "sec": 1000.0,
932 "secs": 1000.0,
933 "seconds": 1000.0,
934 "ms": 1.0,
935 "msec": 1.0,
936 "milliseconds": 1.0,
937 "us": 1e-3,
938 "µs": 1e-3, # MICRO SIGN
939 "μs": 1e-3, # GREEK SMALL LETTER MU — what Tobii writes
940 "microseconds": 1e-3,
941 "ns": 1e-6,
942 "nanoseconds": 1e-6,
943}
944#: Vendor time columns whose unit is in the manual, not the header: Gazepoint's
945#: fixation start/duration and Pupil Labs Core's fixation onset are seconds.
946_SECONDS_WITHOUT_A_SUFFIX = frozenset({"fpogd", "fpogs", "starttimestamp"})
949def time_unit_ms(column) -> float:
950 """Milliseconds per unit of the time column named ``column`` (DATA-40).
952 Read from the header's trailing unit block (``[s]``, ``(ns)``, ``[μs]``) or,
953 for a vendor column that carries none, from its documented unit. Anything
954 else — no block, ``[ms]``, a block that names no time unit — is 1: the app's
955 unit, and the only safe guess.
956 """
957 match = _TRAILING_UNIT.search(str(column))
958 if match:
959 unit = match.group(0).strip().strip("[]()").strip().lower()
960 return _TIME_UNIT_MS.get(unit, 1.0)
961 return 1000.0 if _norm_col(column) in _SECONDS_WITHOUT_A_SUFFIX else 1.0
964def _as_ms(values: pd.Series, column) -> pd.Series:
965 """``values`` of the time column ``column`` converted to milliseconds."""
966 factor = time_unit_ms(column)
967 return values if factor == 1.0 else values * factor
970def trial_mapping_columns(trial_mapping) -> list:
971 """Column list behind a trial mapping — a plain column name or a list of
972 names (the column-mapping UI returns a list when the user composes a
973 unique trial ID from several columns)."""
974 if isinstance(trial_mapping, str):
975 return [trial_mapping]
976 return list(trial_mapping)
979#: Matches an otherwise-integer string with a spurious trailing ``.0`` —
980#: exactly what a whole-number id column becomes once pandas has any reason to
981#: read it as ``float64`` instead of ``int64``.
982_WHOLE_FLOAT_ID = re.compile(r"^(-?\d+)\.0$")
985def stable_id(series: pd.Series) -> pd.Series:
986 """Cast an id column to the string every identity join compares it by.
988 Plain ``.astype(str)`` spells the *same* id two ways when one file's id
989 column is read as whole numbers and another's is read as decimals — and it
990 takes only one blank cell anywhere in an otherwise-integer CSV column to
991 flip the whole column from ``int64`` to ``float64``. A trial's AOI table
992 and its fixations report almost always come from two different exports, so
993 this is not a rare corpus quirk: it is common for exactly one of the two to
994 have that one blank cell the other does not. The trial picker, every
995 metadata join (DATA-20/DATA-29), fixation-to-word assignment, and every
996 filter then compare ``"101"`` against ``"101.0"`` as strings and find no
997 match, even though both name the same trial.
999 Dropping a trailing ``.0`` off an otherwise-integer string is the one
1000 collapse worth making here, and only where the ``.0`` is how *numbers* were
1001 written (round 10): a column read as decimals, a float cell in a mixed
1002 column, or a text column whose every whole number carries ``.0`` (an id
1003 column a script wrote out as floats). A text column that spells some whole
1004 numbers with ``.0`` and some without — ``"1"`` beside ``"1.0"`` — holds
1005 opaque ids, and every spelling in it stays as written. The rule reads the
1006 column's shape, not whether a ``"1"`` happens to sit beside a ``"1.0"``, so
1007 two tables of the same shape decide alike.
1008 """
1009 text = series.astype(str).str.strip()
1010 # A full match and a slice rather than a backreferenced `re.sub`, which
1011 # pandas runs per element in Python (5x slower per million rows).
1012 pointed = text.str.fullmatch(_WHOLE_FLOAT_ID).fillna(False).astype(bool)
1013 if not pointed.any():
1014 return text
1015 collapsed = text.mask(pointed, text.str.slice(stop=-2))
1016 if pd.api.types.is_float_dtype(series):
1017 return collapsed
1018 whole = text.str.fullmatch(r"-?\d+(?:\.0)?").fillna(False).astype(bool)
1019 if bool(pointed[whole].all()):
1020 return collapsed # every whole number written as a float
1021 if series.dtype != object:
1022 return text
1023 # A mixed object column: a real float cell is a number, the rest are text.
1024 # By position, never by label — a concatenated frame repeats its labels.
1025 mask = pointed.to_numpy()
1026 keep = mask.copy()
1027 keep[mask] = [not isinstance(value, float) for value in series.to_numpy()[mask]]
1028 return collapsed.mask(keep, text)
1031_DIGITS_ONLY = re.compile(r"^\d+$")
1034def zero_padding_map(ids: Iterable, reference: Iterable) -> dict[str, str]:
1035 """How ``ids`` would be spelled in ``reference``, when the only thing
1036 keeping the two apart is zero-padding (BUG-59).
1038 One table read ``007`` as text and another read it as the number 7, and
1039 every join between them then matched nothing — words to fixations, a
1040 participant table to the data. Returns ``{"7": "007", …}`` for each id in
1041 ``ids`` whose zero-padded twin is in ``reference``, and ``{}`` whenever that
1042 is not the *only* story: if any id already matches as it is, if either side
1043 has two ids that differ only by padding (``1`` and ``01`` — genuinely
1044 different ids), or if the padded spelling is not all on one side. Nothing is
1045 renamed on a guess.
1046 """
1047 own = {str(v) for v in ids if pd.notna(v)}
1048 other = {str(v) for v in reference if pd.notna(v)}
1049 if not own or not other or own & other:
1050 return {}
1052 def by_value(values: set) -> dict | None:
1053 keyed: dict = {}
1054 for value in values:
1055 if _DIGITS_ONLY.match(value):
1056 key = value.lstrip("0") or "0"
1057 if key in keyed:
1058 return None
1059 keyed[key] = value
1060 return keyed
1062 mine, theirs = by_value(own), by_value(other)
1063 if mine is None or theirs is None:
1064 return {}
1065 shared = mine.keys() & theirs.keys()
1066 if not shared or any(len(mine[k]) >= len(theirs[k]) for k in shared):
1067 return {}
1068 return {mine[k]: theirs[k] for k in shared}
1071#: The separator between the parts of a composite id, and the escape that lets a
1072#: part contain it. See :func:`compose_id`.
1073COMPOSITE_SEPARATOR = "_"
1074_COMPOSITE_ESCAPE = "\\"
1077def _escape_parts(parts: pd.Series) -> pd.Series:
1078 return parts.str.replace(
1079 _COMPOSITE_ESCAPE, _COMPOSITE_ESCAPE * 2, regex=False
1080 ).str.replace(
1081 COMPOSITE_SEPARATOR, _COMPOSITE_ESCAPE + COMPOSITE_SEPARATOR, regex=False
1082 )
1085def compose_id(parts: Iterable) -> str:
1086 r"""One composite id from its parts, such that different parts never give
1087 the same id.
1089 The parts are joined with ``_``. A part that itself contains ``_`` or ``\``
1090 has each one escaped with a ``\`` first, so ``("block_A", "B")`` is
1091 ``block\_A_B`` and ``("block", "A_B")`` is ``block_A\_B`` — before the
1092 escape both were ``block_A_B``, and two readings became one trial. Parts
1093 with neither character, which is almost every id, compose exactly as they
1094 always did (``("p1", "t3")`` is still ``p1_t3``). The encoding is
1095 reversible, which is what makes it injective: :func:`split_composite_id`
1096 reads the parts back. :func:`trial_id_series` is the vectorised form.
1097 """
1098 escaped = _escape_parts(pd.Series([str(part) for part in parts], dtype=object))
1099 return COMPOSITE_SEPARATOR.join(escaped)
1102def split_composite_id(value: str) -> list[str]:
1103 """The parts :func:`compose_id` joined into ``value``."""
1104 parts: list[str] = []
1105 current: list[str] = []
1106 chars = iter(str(value))
1107 for char in chars:
1108 if char == _COMPOSITE_ESCAPE:
1109 current.append(next(chars, _COMPOSITE_ESCAPE))
1110 elif char == COMPOSITE_SEPARATOR:
1111 parts.append("".join(current))
1112 current = []
1113 else:
1114 current.append(char)
1115 parts.append("".join(current))
1116 return parts
1119def legacy_composite_id(value: str) -> str:
1120 """How a composite id was spelled before :func:`compose_id` escaped its
1121 parts — plainly joined with ``_``. The same as ``value`` unless one of its
1122 parts held a ``_`` or a backslash."""
1123 value = str(value)
1124 if _COMPOSITE_ESCAPE not in value:
1125 return value
1126 return COMPOSITE_SEPARATOR.join(split_composite_id(value))
1129def composite_respelling_map(ids: Iterable, reference: Iterable) -> dict[str, str]:
1130 r"""How each id in ``ids`` is spelled in ``reference``, when the only thing
1131 keeping them apart is the escaping :func:`compose_id` added to composite ids.
1133 An id saved before it — in a stored dataset, an annotations file, a link —
1134 spells a composite id whose parts contain ``_`` without the escapes; one
1135 composed since spells it with them. Returns ``{"block_A_B": "block\_A_B"}``
1136 (or the reverse), for the ids of ``ids`` that ``reference`` lacks. An old
1137 spelling two current ids share is left out: that old id named both readings
1138 at once, and nothing here picks one. Nothing is renamed on a guess, as with
1139 :func:`zero_padding_map`.
1140 """
1141 own = {str(v) for v in ids if pd.notna(v)}
1142 other = {str(v) for v in reference if pd.notna(v)}
1143 missing = own - other
1144 if not missing or not other:
1145 return {}
1146 by_legacy: dict[str, str | None] = {}
1147 for value in other:
1148 legacy = legacy_composite_id(value)
1149 if legacy != value:
1150 by_legacy[legacy] = None if legacy in by_legacy else value
1151 mapping: dict[str, str] = {}
1152 for value in missing:
1153 current = by_legacy.get(value)
1154 if current is not None:
1155 mapping[value] = current
1156 continue
1157 legacy = legacy_composite_id(value)
1158 if legacy != value and legacy in other:
1159 mapping[value] = legacy
1160 # Two ids landing on one would merge them — leave both alone.
1161 landed: dict[str, int] = {}
1162 for target in mapping.values():
1163 landed[target] = landed.get(target, 0) + 1
1164 return {k: v for k, v in mapping.items() if landed[v] == 1}
1167def respell_reading(participant, trial, readings: Iterable) -> tuple[str, str]:
1168 """``(participant, trial)`` spelled the way ``readings`` — ``(participant,
1169 trial)`` pairs — spell them, through :func:`composite_respelling_map`.
1171 For an id saved before composite ids escaped a ``_`` inside a part: a link,
1172 an annotations file or a script. Each half is respelled only when it is
1173 missing as given and its other spelling is unambiguous; otherwise it comes
1174 back unchanged and the caller's own "not found" applies.
1175 """
1176 pid, tid = str(participant), str(trial)
1177 pairs = (
1178 readings
1179 if isinstance(readings, frozenset) # already strings, e.g. a trial set
1180 else frozenset((str(p), str(t)) for p, t in readings)
1181 )
1182 if (pid, tid) in pairs:
1183 return pid, tid
1184 pid = composite_respelling_map([pid], {p for p, _ in pairs}).get(pid, pid)
1185 own = {t for p, t in pairs if p == pid} or {t for _, t in pairs}
1186 return pid, composite_respelling_map([tid], own).get(tid, tid)
1189def trial_id_series(source: pd.DataFrame, trial_mapping) -> pd.Series:
1190 """Trial-id values for a single-column or composite (multi-column) mapping.
1192 A multi-column mapping builds a unique trial ID on the fly from the
1193 columns' string values with :func:`compose_id` — joined with ``_``, a ``_``
1194 or backslash inside a part escaped, so two different tuples never share an
1195 id — for datasets that ship no precomputed unique-trial column (e.g.
1196 OneStop-style participant + paragraph + repeated-reading). Each component
1197 is passed through :func:`stable_id` first, so a composite id cannot inherit
1198 a ``.0`` from one of its parts.
1199 """
1200 cols = trial_mapping_columns(trial_mapping)
1201 if len(cols) == 1:
1202 return stable_id(source[cols[0]])
1203 escaped = [_escape_parts(stable_id(source[c])) for c in cols]
1204 return escaped[0].str.cat(escaped[1:], sep=COMPOSITE_SEPARATOR)
1207def _preserve_composite_columns(
1208 df: pd.DataFrame, source: pd.DataFrame, trial_mapping
1209) -> pd.DataFrame:
1210 """Carry a composite mapping's source columns into the normalized frame
1211 under their original names.
1213 A multi-column trial mapping gets joined into a single opaque ``trial_id``
1214 (e.g. ``2_1_1_Ele_l37_1129_False``). Keeping the individual component columns
1215 is what lets the trial chips spell that id back out part by part
1216 (``tabs._render_trial_condition_chips``). It no longer changes how the trial
1217 is *picked* — BUG-23 made the picker the same for every mapping.
1218 No-op for single-column mappings.
1219 Rows are 1:1 with ``source`` here (no filtering in the composite path), so a
1220 positional copy stays aligned."""
1221 cols = trial_mapping_columns(trial_mapping)
1222 if len(cols) < 2:
1223 return df
1224 for col in cols:
1225 if col not in df.columns and col in source.columns:
1226 df[col] = source[col].to_numpy()
1227 return df
1230# Candidate column names checked during auto-inference. Centralised so the
1231# proposal step and the override UI share the same defaults. Matching is case-
1232# and separator-insensitive (see ``pick_column``), so these list only *distinct*
1233# conventions — no ALL_CAPS / snake_case twins of the same name needed.
1234PARTICIPANT_CANDIDATES = [
1235 "participant_id",
1236 "unique_participant_id", # EyeGenBench's own harmonized column name (DATA-27)
1237 "subject_id",
1238 "participant", # also SMI BeGaze's event export
1239 "participant_name", # Tobii Pro Lab `Participant name` / Tobii Studio `ParticipantName`
1240 "recording_session_label",
1241 "reader_id",
1242 "USER", # Gazepoint
1243 "recording_name", # Tobii Pro Lab — one recording per participant session
1244 "recording_id", # Pupil Labs Neon
1245]
1246TRIAL_CANDIDATES = [
1247 "unique_trial_id",
1248 "trial_id",
1249 "unique_paragraph_id",
1250 "paragraph_id",
1251 "text_id",
1252 "trial",
1253 "trial_index",
1254 "trial_number", # SMI BeGaze event export `Trial Number`
1255 # One stimulus per trial is how Tobii, SMI and Gazepoint exports are
1256 # shaped, so the stimulus name is the trial key when nothing above exists.
1257 "presented_stimulus_name", # Tobii Pro Lab
1258 "media_name", # Tobii Studio `MediaName` / Gazepoint `MEDIA_NAME`
1259 "stimulus", # SMI BeGaze
1260]
1261SCREEN_ID_CANDIDATES = ["screen_id", "part_id", "page_id", "screen", "page", "part"]
1262SCREEN_INDEX_CANDIDATES = [
1263 "screen_index",
1264 "part_index",
1265 "page_index",
1266 "screen_order",
1267 "page_number",
1268]
1269SCREEN_TIMESTAMP_CANDIDATES = [
1270 "screen_timestamp_ms",
1271 "screen_time_ms",
1272 "page_timestamp_ms",
1273 "local_timestamp_ms",
1274]
1275SCREEN_FIXATION_ID_CANDIDATES = [
1276 "screen_fixation_id",
1277 "screen_fixation_index",
1278 "page_fixation_id",
1279 "local_fixation_id",
1280]
1281CANVAS_WIDTH_CANDIDATES = ["canvas_width", "screen_width", "monitor_width"]
1282CANVAS_HEIGHT_CANDIDATES = ["canvas_height", "screen_height", "monitor_height"]
1283# Source column names that identify which *text* (passage) a row belongs to.
1284# Output canonical column is `text_id` (was `paragraph_id`); the source names stay
1285# as the real-world conventions so auto-detection keeps working.
1286TEXT_ID_CANDIDATES = [
1287 "unique_paragraph_id",
1288 "paragraph_id",
1289 "unique_text_id",
1290 "text_id",
1291 "presented_stimulus_name", # Tobii Pro Lab — the stimulus *is* the text
1292 "media_name", # Tobii Studio / Gazepoint
1293 "stimulus", # SMI BeGaze
1294 # #374 F13: an EyeLink export's own item column (`item`, `ITEM_ID`, …) —
1295 # last, so a named text/paragraph column still wins.
1296 "item",
1297 "item_id",
1298]
1299TEXT_CANDIDATES = [
1300 "text",
1301 "IA_LABEL",
1302 "label",
1303 "word",
1304 "content",
1305 "token",
1306]
1307# `word_idx` / `char_idx` are MultiplEYE's word- and character-level indices
1308# (word_idx first so word-level boxes win over per-character ones).
1309WORD_ID_CANDIDATES = [
1310 "word_id",
1311 "IA_ID",
1312 "ia_index",
1313 "word_index",
1314 "aoi",
1315 "word_idx",
1316 "char_idx",
1317]
1318LINE_CANDIDATES = ["line_idx", "line", "line_index", "IA_LINE_ID"]
1320# `top_left_x` / `top_left_y` are MultiplEYE's box origin (paired with width/height).
1321WORD_X_CANDIDATES = ["x", "left", "top_left_x"]
1322WORD_Y_CANDIDATES = ["y", "top", "top_left_y"]
1323WORD_WIDTH_CANDIDATES = ["width"]
1324WORD_HEIGHT_CANDIDATES = ["height"]
1325WORD_LEFT_CANDIDATES = ["IA_LEFT", "left", "start_x", "top_left_x"]
1326WORD_RIGHT_CANDIDATES = ["IA_RIGHT", "right", "end_x"]
1327WORD_TOP_CANDIDATES = ["IA_TOP", "top", "start_y", "top_left_y"]
1328WORD_BOTTOM_CANDIDATES = ["IA_BOTTOM", "bottom", "end_y"]
1330# `location_x` / `location_y` are MultiplEYE's fixation pixel coordinates.
1331# Tobii Pro Lab: `Fixation point X [DACS px]`; Tobii Studio: `FixationPointX
1332# (MCSpx)`; SMI BeGaze: `Position X [px]` / `Fixation Position X`; Pupil Labs
1333# Neon: `fixation x [px]`. Gazepoint's FPOGX and Pupil Core's `norm_pos_x` are
1334# screen *fractions* (0–1), not pixels — matched here so the column is found,
1335# and reported as fractions by `screen_fraction_issues` (DATA-40), which is as
1336# far as the load can go without knowing the screen size.
1337FIX_X_CANDIDATES = [
1338 "x",
1339 "CURRENT_FIX_X",
1340 "FPOGX",
1341 "location_x",
1342 "fixation_point_x",
1343 "fixation_x",
1344 "position_x",
1345 "fixation_position_x",
1346 "norm_pos_x",
1347]
1348FIX_Y_CANDIDATES = [
1349 "y",
1350 "CURRENT_FIX_Y",
1351 "FPOGY",
1352 "location_y",
1353 "fixation_point_y",
1354 "fixation_y",
1355 "position_y",
1356 "fixation_position_y",
1357 "norm_pos_y",
1358]
1359FIX_DURATION_CANDIDATES = [
1360 "duration_ms",
1361 "CURRENT_FIX_DURATION",
1362 "CURRENT_FIX_LEN",
1363 "duration", # also SMI `Duration [ms]`, Pupil Labs `duration [ms]` / `duration`
1364 "fixation_duration", # SMI BeGaze `Fixation Duration [ms]`
1365 "fix_duration", # EyeGenBench's own harmonized column name (DATA-27)
1366 "eye_movement_event_duration", # Tobii Pro Lab (current name)
1367 "gaze_event_duration", # Tobii Pro Lab (older) / Tobii Studio `GazeEventDuration`
1368 "FPOGD", # Gazepoint — seconds, not ms (`time_unit_ms` converts, DATA-40)
1369]
1370FIX_TIMESTAMP_CANDIDATES = [
1371 "timestamp_ms",
1372 "CURRENT_FIX_START",
1373 "CURRENT_FIX_START_TIME",
1374 "CURRENT_FIX_TIME",
1375 "CURRENT_FIX_ONSET",
1376 "onset", # MultiplEYE fixation onset (ms)
1377 "fixation_start", # SMI BeGaze `Fixation Start [ms]`
1378 "event_start_trial_time", # SMI BeGaze `Event Start Trial Time [ms]`
1379 "start_timestamp", # Pupil Labs Core `start_timestamp` / Neon `start timestamp [ns]`
1380 "recording_timestamp", # Tobii Pro Lab `Recording timestamp [ms]` / Studio `RecordingTimestamp`
1381 "FPOGS", # Gazepoint — seconds
1382]
1383FIX_FIXATION_ID_CANDIDATES = [
1384 "fixation_id", # also Pupil Labs Neon `fixation id`
1385 "CURRENT_FIX_INDEX",
1386 "CURRENT_FIX_NUM",
1387 "fixation_index", # also Tobii Studio `FixationIndex`
1388 "fix_index", # EyeGenBench's own harmonized column name (DATA-27)
1389 "eye_movement_type_index", # Tobii Pro Lab
1390 "FPOGID", # Gazepoint
1391]
1392FIX_WORD_ID_CANDIDATES = [
1393 "word_id",
1394 "IA_ID",
1395 "ia_index", # EyeGenBench's own harmonized column name (DATA-27)
1396 "CURRENT_FIX_INTEREST_AREA_ID",
1397 "CURRENT_FIX_INTEREST_AREA_INDEX",
1398 "word_index_in_text",
1399 "word_index",
1400 "word_idx", # MultiplEYE word index (resets per page)
1401 "char_idx", # MultiplEYE character index
1402]
1403# Tobii Pro Lab `Gaze point X [DACS px]`, Tobii Studio `GazePointX`, Pupil Labs
1404# Neon `gaze x [px]`, SMI raw IDF `L POR X [px]` (left eye first; the right eye
1405# is the fallback when only it was recorded).
1406RAW_GAZE_X_CANDIDATES = ["x", "FPOGX", "gaze_x", "gaze_point_x", "l_por_x", "r_por_x"]
1407RAW_GAZE_Y_CANDIDATES = ["y", "FPOGY", "gaze_y", "gaze_point_y", "l_por_y", "r_por_y"]
1408RAW_GAZE_TIMESTAMP_CANDIDATES = [
1409 "timestamp", # also Pupil Labs Neon `timestamp [ns]`
1410 "time", # also SMI raw IDF `Time`, Gazepoint `TIME`
1411 "ms",
1412 "timestamp_ms",
1413 "time_ms",
1414 "recording_timestamp", # Tobii Pro Lab / Tobii Studio
1415]
1418_BOX_EDGES = ("left", "right", "top", "bottom")
1421def _pick_box_edge_set(words: pd.DataFrame) -> dict[str, str] | None:
1422 """The four word-box edge columns, resolved as one set (DATA-57).
1424 ``pick_column`` looks at each edge on its own, and its second pass accepts a
1425 prefixed or suffixed name (``LEFT_px``, ``aoi_left``) only when it is the
1426 *only* column carrying that token. An AOI export routinely carries two box
1427 encodings side by side — EyeLink's ``LEFT_px`` … ``BOTTOM_px`` next to a
1428 derived ``aoi_left`` … ``aoi_bottom`` — so every edge was ambiguous and the
1429 whole box landed in the manual step. The edges are not independent: they
1430 share an affix. So each column naming exactly one edge is keyed by the rest
1431 of its name (``*_px``, ``aoi_*``), and a key that covers all four edges is a
1432 set. The set that comes **last** in the table wins: a derived box is
1433 usually appended after the one the export shipped with, and the columns a
1434 lab adds later are the ones it means (``aoi_left`` … then ``LEFT_px`` …
1435 picks ``LEFT_px``).
1437 Returns ``{edge: column}`` plus the shared affix under ``"affix"`` (for
1438 ``_affix_sibling``), or ``None`` when no complete set exists."""
1439 groups: dict[tuple[str, ...], dict[str, str]] = {}
1440 order: dict[tuple[str, ...], int] = {}
1441 for pos, col in enumerate(words.columns):
1442 tokens = _col_tokens(col)
1443 edges = [tok for tok in tokens if tok in _BOX_EDGES]
1444 if len(edges) != 1:
1445 continue
1446 affix = tuple("*" if tok == edges[0] else tok for tok in tokens)
1447 group = groups.setdefault(affix, {})
1448 if edges[0] not in group:
1449 group[edges[0]] = col
1450 order.setdefault(affix, pos)
1451 complete = [a for a, g in groups.items() if len(g) == len(_BOX_EDGES)]
1452 if not complete:
1453 return None
1454 affix = max(complete, key=order.__getitem__)
1455 return {**groups[affix], "affix": affix}
1458def _affix_sibling(
1459 words: pd.DataFrame, affix: tuple[str, ...], token: str
1460) -> str | None:
1461 """The column named like an edge set's affix with ``token`` in the edge's
1462 place — ``aoi_width`` beside ``aoi_left`` … ``aoi_bottom``."""
1463 want = [token if tok == "*" else tok for tok in affix]
1464 return next((col for col in words.columns if _col_tokens(col) == want), None)
1467#: AN-32 — the per-AOI reading measures a dataset *brings*: the Corpus Analysis
1468#: page shows these and computes none of them. Each is an optional AOI-table
1469#: field, ``(schema key, canonical column, short label, full name, kind,
1470#: candidates)``; the candidates lead with EyeLink Data Viewer's interest-area
1471#: report names (``IA_*``), then the plain spellings other exports use. The
1472#: canonical columns are ``aggregation.MEASURES``' own, so a mapped measure is
1473#: the one the page's pickers offer.
1474READING_MEASURE_FIELDS: tuple[tuple[str, str, str, str, str, tuple[str, ...]], ...] = (
1475 (
1476 "measure_tfd",
1477 "total_fixation_duration_ms",
1478 "TFD",
1479 "Total fixation duration (dwell time), ms",
1480 "numeric",
1481 ("IA_DWELL_TIME", "total_fixation_duration", "TFD", "dwell_time"),
1482 ),
1483 (
1484 "measure_ffd",
1485 "first_fixation_ms",
1486 "FFD",
1487 "First fixation duration, ms",
1488 "numeric",
1489 ("IA_FIRST_FIXATION_DURATION", "first_fixation_duration", "FFD"),
1490 ),
1491 (
1492 "measure_fprt",
1493 "first_pass_gaze_duration_ms",
1494 "FPRT",
1495 "First-pass reading time (gaze duration), ms",
1496 "numeric",
1497 (
1498 "IA_FIRST_RUN_DWELL_TIME",
1499 "first_pass_gaze_duration",
1500 "gaze_duration",
1501 "FPRT",
1502 ),
1503 ),
1504 (
1505 "measure_rpd",
1506 "regression_path_duration_ms",
1507 "RPD",
1508 "Regression-path (go-past) duration, ms",
1509 "numeric",
1510 ("IA_REGRESSION_PATH_DURATION", "regression_path_duration", "go_past", "RPD"),
1511 ),
1512 (
1513 "measure_second_pass",
1514 "second_pass_duration_ms",
1515 "2nd pass",
1516 "Second-pass duration, ms",
1517 "numeric",
1518 ("IA_SECOND_RUN_DWELL_TIME", "second_pass_duration"),
1519 ),
1520 (
1521 "measure_single_fix",
1522 "single_fixation_duration_ms",
1523 "Single fix.",
1524 "Single-fixation duration, ms",
1525 "numeric",
1526 ("IA_SINGLE_FIXATION_DURATION", "single_fixation_duration", "SFD"),
1527 ),
1528 (
1529 "measure_nfix",
1530 "n_fixations",
1531 "Fix. count",
1532 "Number of fixations on the word",
1533 "numeric",
1534 ("IA_FIXATION_COUNT", "fixation_count", "n_fixations"),
1535 ),
1536 (
1537 "measure_skip",
1538 "skip_flag",
1539 "Skip",
1540 "Skipped in first pass (0/1)",
1541 "boolean",
1542 ("IA_SKIP", "skip_flag", "skipped"),
1543 ),
1544 (
1545 "measure_reg_in",
1546 "regression_in_flag",
1547 "Reg. in",
1548 "Regressed into (0/1)",
1549 "boolean",
1550 ("IA_REGRESSION_IN", "regression_in_flag"),
1551 ),
1552 (
1553 "measure_reg_out",
1554 "regression_out_flag",
1555 "Reg. out",
1556 "Regressed out of (0/1)",
1557 "boolean",
1558 ("IA_REGRESSION_OUT", "regression_out_flag"),
1559 ),
1560 (
1561 "measure_reg_in_count",
1562 "number_of_regressions_in",
1563 "Reg. in (n)",
1564 "Number of regressions into the word",
1565 "numeric",
1566 ("IA_REGRESSION_IN_COUNT", "number_of_regressions_in", "regression_in_count"),
1567 ),
1568 (
1569 "measure_landing_position",
1570 "initial_landing_position",
1571 "Landing pos.",
1572 "Initial landing position, letters",
1573 "numeric",
1574 ("IA_FIRST_FIXATION_LANDING_POSITION", "initial_landing_position"),
1575 ),
1576 (
1577 "measure_landing_distance",
1578 "initial_landing_distance",
1579 "Landing dist.",
1580 "Centered initial landing distance, letters",
1581 "numeric",
1582 ("initial_landing_distance", "landing_distance"),
1583 ),
1584)
1585READING_MEASURE_KEYS: tuple[str, ...] = tuple(f[0] for f in READING_MEASURE_FIELDS)
1586READING_MEASURE_COLUMNS: tuple[str, ...] = tuple(f[1] for f in READING_MEASURE_FIELDS)
1589def brought_reading_measures(words: pd.DataFrame | None) -> list[str]:
1590 """The reading-measure columns ``words`` carries — what the dataset brought.
1592 AN-32 / EXP-23: the Corpus Analysis page and the export show these and
1593 compute none, so this only *reads* the frame; an empty list means the
1594 dataset has no reading measures."""
1595 if words is None:
1596 return []
1597 return [column for column in READING_MEASURE_COLUMNS if column in words.columns]
1600def _apply_reading_measures(
1601 df: pd.DataFrame, source: pd.DataFrame, schema: dict
1602) -> None:
1603 """Write the mapped reading measures onto ``df`` (AN-32).
1605 A schema that names a measure key decides that measure outright: mapped, the
1606 column is copied under its canonical name (and wins over the optional-field
1607 passthrough); cleared, the canonical column is removed even if the
1608 passthrough carried it — "this dataset has no TFD" has to stay true. A
1609 schema without measure keys (a dataset stored before AN-32) is left as the
1610 passthrough made it."""
1611 for key, canonical, _label, _name, kind, _candidates in READING_MEASURE_FIELDS:
1612 if key not in schema:
1613 continue
1614 column = schema.get(key)
1615 if column and column in source.columns:
1616 values = source[column]
1617 df[canonical] = (
1618 coerce_measure_flag(values) if kind == "boolean" else _to_number(values)
1619 )
1620 elif canonical in df.columns:
1621 del df[canonical]
1624#: The duration measures a word nobody fixated does not have (BUG-63). Total
1625#: fixation duration is not one of them: the word was read past and got 0 ms.
1626UNFIXATED_BLANK_MEASURES: tuple[str, ...] = (
1627 "first_fixation_ms",
1628 "first_pass_gaze_duration_ms",
1629 "regression_path_duration_ms",
1630 "single_fixation_duration_ms",
1631)
1634def _blank_unfixated_measures(df: pd.DataFrame) -> None:
1635 """Blank the imported FFD / FPRT / RPD / single-fixation duration of a word
1636 nobody fixated (BUG-63's rule, on the imported path — #374 F1).
1638 An EyeLink IA report writes ``0`` there, and every mean then counted a
1639 skipped word as a 0 ms fixation. "Never fixated" is a fixation count of 0;
1640 where no count is mapped (or the cell is blank), a total fixation duration
1641 of 0 says the same. Total fixation duration itself keeps its 0.
1643 The mirror image, as on the computed path: second pass is "fewer than two
1644 runs ⇒ 0", but an imported IA_SECOND_RUN_DWELL_TIME leaves those cells
1645 blank, so its mean would cover only re-read words. Where the fixation count
1646 is known, a blank second pass becomes 0."""
1647 if "second_pass_duration_ms" in df.columns and "n_fixations" in df.columns:
1648 known = pd.to_numeric(df["n_fixations"], errors="coerce").notna()
1649 second = pd.to_numeric(df["second_pass_duration_ms"], errors="coerce")
1650 df["second_pass_duration_ms"] = second.mask(known & second.isna(), 0.0)
1651 present = [column for column in UNFIXATED_BLANK_MEASURES if column in df.columns]
1652 if not present:
1653 return
1654 unfixated = pd.Series(False, index=df.index)
1655 count = (
1656 pd.to_numeric(df["n_fixations"], errors="coerce")
1657 if "n_fixations" in df.columns
1658 else pd.Series(np.nan, index=df.index)
1659 )
1660 unfixated |= count.eq(0)
1661 if "total_fixation_duration_ms" in df.columns:
1662 total = pd.to_numeric(df["total_fixation_duration_ms"], errors="coerce")
1663 unfixated |= count.isna() & total.eq(0)
1664 if not unfixated.any():
1665 return
1666 for column in present:
1667 df[column] = pd.to_numeric(df[column], errors="coerce").mask(unfixated)
1670def propose_word_schema(words: pd.DataFrame) -> dict[str, str | None]:
1671 """Return a candidate column mapping for words/IA data without erroring."""
1672 schema = _propose_word_schema_by_field(words)
1673 # BUG-99: a box field is a number, so a column with no number in it is not
1674 # one, whatever its name says. OneStop's IA report carries `TOP_LEFT`, the
1675 # box corner as the text `(368,186)`; its tokens matched both the x and the
1676 # y list, so the load proposed it for both and warned about every cell.
1677 for key in _BOX_FIELDS:
1678 if schema[key] and _holds_no_number(words[schema[key]]):
1679 schema[key] = None
1680 # AN-32: every reading measure is proposed too, from its known names.
1681 # The app's own canonical name is a candidate too, right after EyeLink's,
1682 # so a table the app itself wrote (an exported words.csv, a normalized
1683 # frame handed to the API) keeps its measures instead of having the key
1684 # proposed empty — which `_apply_reading_measures` reads as "absent".
1685 for key, column, _label, _name, _kind, candidates in READING_MEASURE_FIELDS:
1686 schema[key] = pick_column(words, (candidates[0], column, *candidates[1:]))
1687 if all(schema[edge] for edge in _BOX_EDGES):
1688 return schema
1689 edge_set = _pick_box_edge_set(words)
1690 if edge_set is None:
1691 return schema
1692 affix = edge_set.pop("affix")
1693 schema.update(edge_set)
1694 # The origin + size fields follow the same set, so the two encodings the
1695 # mapping screen offers describe one box rather than two.
1696 schema["x"] = schema["x"] or edge_set["left"]
1697 schema["y"] = schema["y"] or edge_set["top"]
1698 for size in ("width", "height"):
1699 schema[size] = _affix_sibling(words, affix, size)
1700 return schema
1703_BOX_FIELDS = ("x", "y", "width", "height", *_BOX_EDGES)
1706def _holds_no_number(values: pd.Series) -> bool:
1707 """Whether ``values`` has filled cells and none of them reads as a number.
1709 A header-only frame (the mapping screen proposes from one) has no cells and
1710 so is never ruled out; a numeric column with a few bad cells is still a
1711 numeric column, and stays proposed for :func:`numeric_parse_issues` to
1712 report."""
1713 if pd.api.types.is_numeric_dtype(values) or pd.api.types.is_bool_dtype(values):
1714 return False
1715 filled = _filled_cells(values)
1716 return not filled.empty and _to_number(filled).isna().all()
1719def _propose_word_schema_by_field(words: pd.DataFrame) -> dict[str, str | None]:
1720 return dict(
1721 participant=pick_column(words, PARTICIPANT_CANDIDATES),
1722 trial=pick_column(words, TRIAL_CANDIDATES),
1723 screen_id=pick_column(words, SCREEN_ID_CANDIDATES),
1724 screen_index=pick_column(words, SCREEN_INDEX_CANDIDATES),
1725 canvas_width=pick_column(words, CANVAS_WIDTH_CANDIDATES),
1726 canvas_height=pick_column(words, CANVAS_HEIGHT_CANDIDATES),
1727 text_id=pick_column(words, TEXT_ID_CANDIDATES),
1728 word_id=pick_column(words, WORD_ID_CANDIDATES),
1729 text=pick_column(words, TEXT_CANDIDATES),
1730 line=pick_column(words, LINE_CANDIDATES),
1731 x=pick_column(words, WORD_X_CANDIDATES),
1732 y=pick_column(words, WORD_Y_CANDIDATES),
1733 width=pick_column(words, WORD_WIDTH_CANDIDATES),
1734 height=pick_column(words, WORD_HEIGHT_CANDIDATES),
1735 left=pick_column(words, WORD_LEFT_CANDIDATES),
1736 right=pick_column(words, WORD_RIGHT_CANDIDATES),
1737 top=pick_column(words, WORD_TOP_CANDIDATES),
1738 bottom=pick_column(words, WORD_BOTTOM_CANDIDATES),
1739 )
1742def propose_fix_schema(fixations: pd.DataFrame) -> dict[str, str | None]:
1743 """Return a candidate column mapping for fixations data without erroring.
1745 pass_index / saccade_type / saccade_amplitude / eye are not schema fields —
1746 they're auto-detected and kept via ``FIX_OPTIONAL_FIELDS`` (and offered under
1747 *fields to keep*), so they're not proposed here."""
1748 return dict(
1749 participant=pick_column(fixations, PARTICIPANT_CANDIDATES),
1750 trial=pick_column(fixations, TRIAL_CANDIDATES),
1751 screen_id=pick_column(fixations, SCREEN_ID_CANDIDATES),
1752 screen_index=pick_column(fixations, SCREEN_INDEX_CANDIDATES),
1753 screen_timestamp=pick_column(fixations, SCREEN_TIMESTAMP_CANDIDATES),
1754 screen_fixation_id=pick_column(fixations, SCREEN_FIXATION_ID_CANDIDATES),
1755 canvas_width=pick_column(fixations, CANVAS_WIDTH_CANDIDATES),
1756 canvas_height=pick_column(fixations, CANVAS_HEIGHT_CANDIDATES),
1757 text_id=pick_column(fixations, TEXT_ID_CANDIDATES),
1758 fixation_id=pick_column(fixations, FIX_FIXATION_ID_CANDIDATES),
1759 timestamp=pick_column(fixations, FIX_TIMESTAMP_CANDIDATES),
1760 duration=pick_column(fixations, FIX_DURATION_CANDIDATES),
1761 x=pick_column(fixations, FIX_X_CANDIDATES),
1762 y=pick_column(fixations, FIX_Y_CANDIDATES),
1763 word_id=pick_column(fixations, FIX_WORD_ID_CANDIDATES),
1764 )
1767def propose_raw_gaze_schema(raw_gaze: pd.DataFrame) -> dict[str, str | None]:
1768 """Return a candidate column mapping for raw gaze data without erroring."""
1769 return dict(
1770 participant=pick_column(raw_gaze, PARTICIPANT_CANDIDATES),
1771 trial=pick_column(raw_gaze, TRIAL_CANDIDATES),
1772 screen_id=pick_column(raw_gaze, SCREEN_ID_CANDIDATES),
1773 screen_index=pick_column(raw_gaze, SCREEN_INDEX_CANDIDATES),
1774 text_id=pick_column(raw_gaze, TEXT_ID_CANDIDATES),
1775 word_id=pick_column(raw_gaze, FIX_WORD_ID_CANDIDATES),
1776 text=pick_column(raw_gaze, TEXT_CANDIDATES),
1777 x=pick_column(raw_gaze, RAW_GAZE_X_CANDIDATES),
1778 y=pick_column(raw_gaze, RAW_GAZE_Y_CANDIDATES),
1779 timestamp=pick_column(raw_gaze, RAW_GAZE_TIMESTAMP_CANDIDATES),
1780 )
1783def validate_word_schema(schema: dict[str, str | None]) -> list:
1784 """Return a list of human-readable problems with a words/IA schema.
1786 Participant ID is optional: word/AoI tables without one are treated as
1787 stimulus-level (one row per word per *text*, not per reading) and are
1788 broadcast across the participants found in the fixations — see
1789 ``broadcast_stimulus_words``."""
1790 problems = []
1791 for key, label in [
1792 ("trial", "Trial ID"),
1793 ("word_id", "Word/IA ID"),
1794 ]:
1795 if not schema.get(key):
1796 problems.append(f"missing {label}")
1797 has_xywh = all(schema.get(k) for k in ["x", "y", "width", "height"])
1798 has_box = all(schema.get(k) for k in ["left", "right", "top", "bottom"])
1799 if not has_xywh and not has_box:
1800 problems.append("missing Word box (its edges, or x/y with width and height)")
1801 return problems
1804def validate_fix_schema(schema: dict[str, str | None]) -> list:
1805 """Return a list of human-readable problems with a fixations schema.
1807 X/Y coordinates are optional when a Word/IA ID is mapped: AOI-sequence
1808 datasets (fixations recorded as "which word", not "which pixel") get
1809 coordinates from the matching word-box centers — see
1810 ``fill_fixation_xy_from_words``."""
1811 problems = []
1812 # Participant is optional — a dataset without it is treated as a single
1813 # anonymous reader (see SYNTHETIC_PARTICIPANT).
1814 for key, label in [
1815 ("trial", "Trial ID"),
1816 ("duration", "Duration"),
1817 ]:
1818 if not schema.get(key):
1819 problems.append(f"missing {label}")
1820 has_xy = schema.get("x") and schema.get("y")
1821 if not has_xy and not schema.get("word_id"):
1822 problems.append(
1823 "missing X and Y (or a Word/IA ID, to place each fixation at its "
1824 "word's center)"
1825 )
1826 return problems
1829def validate_raw_gaze_schema(schema: dict[str, str | None]) -> list:
1830 """Return a list of human-readable problems with a raw gaze schema."""
1831 problems = []
1832 # Participant optional — single anonymous reader when absent.
1833 for key, label in [
1834 ("trial", "Trial ID"),
1835 ("x", "X"),
1836 ("y", "Y"),
1837 ]:
1838 if not schema.get(key):
1839 problems.append(f"missing {label}")
1840 return problems
1843# Column added when concatenating several files (multi-file upload, glob, or a
1844# multi-member zip): the source file's stem. Lets datasets that key metadata in
1845# the *filename* (one file per participant and/or per text, e.g. PoTeC's
1846# `reader0_b0_scanpath.tsv`) recover it after concatenation — map it as (part
1847# of) the Trial/Participant ID.
1848SOURCE_FILE_COLUMN = "source_file"
1850# Prefix for the positional columns split out of `source_file` by
1851# `split_source_file` (file_part_1, file_part_2, …).
1852FILE_PART_PREFIX = "file_part_"
1854TablesInput = str | os.PathLike | object | list
1857def split_source_file(
1858 df: pd.DataFrame,
1859 *,
1860 delimiter: str = "_",
1861 column: str = SOURCE_FILE_COLUMN,
1862 prefix: str = FILE_PART_PREFIX,
1863) -> pd.DataFrame:
1864 """Split a ``source_file`` column into positional ``file_part_N`` columns.
1866 Lets the upload wizard derive a trial / participant id from a structured
1867 filename when no data column carries it — e.g. ``reader0_b0_scanpath`` split
1868 on ``_`` yields ``file_part_1=reader0``, ``file_part_2=b0``,
1869 ``file_part_3=scanpath`` (the user then maps the relevant part(s), composing
1870 several if needed). Returns ``df`` unchanged if ``column`` is absent or
1871 ``delimiter`` is empty. Rows with fewer parts get empty strings for the
1872 missing tail, so every row has the same part columns."""
1873 if column not in df.columns or not delimiter:
1874 return df
1875 parts = (
1876 df[column].astype(str).str.split(delimiter, expand=True, regex=False).fillna("")
1877 )
1878 df = df.copy()
1879 for i in range(parts.shape[1]):
1880 df[f"{prefix}{i + 1}"] = parts[i].to_numpy()
1881 return df
1884def extract_columns_from_source_file(
1885 df: pd.DataFrame,
1886 pattern: str,
1887 *,
1888 column: str = SOURCE_FILE_COLUMN,
1889 lowercase: bool = False,
1890) -> pd.DataFrame:
1891 """Add one column per *named group* of a regex applied to ``source_file``.
1893 Sibling of :func:`split_source_file` for filenames whose fields are
1894 positionally irregular — varying-length parts or an optional prefix make a
1895 fixed delimiter split unreliable. A regex with named groups, e.g.
1896 ``r"(?P<session>\\d+_\\w+_ET\\d)_.*_(?P<stimulus>.+)_scanpath"``, extracts each
1897 group into its own column the wizard can then map as a trial / participant id.
1898 ``lowercase`` folds the captured values (useful when one table names a field
1899 CamelCase and another lowercase). No-op (returns ``df`` unchanged) when
1900 ``column`` is absent, ``pattern`` is empty / uncompilable, or it declares no
1901 named groups; rows that don't match get NaN. A named group that collides with
1902 an existing column is **skipped** (the real data wins) — see
1903 :func:`source_file_regex_collisions` to surface those in a UI."""
1904 if not pattern or column not in df.columns:
1905 return df
1906 try:
1907 compiled = re.compile(pattern)
1908 except re.error:
1909 return df
1910 if not compiled.groupindex:
1911 return df
1912 extracted = df[column].astype(str).str.extract(compiled)
1913 df = df.copy()
1914 for group in compiled.groupindex: # named groups only
1915 if group in df.columns: # don't clobber an existing data column
1916 continue
1917 values = extracted[group]
1918 if lowercase:
1919 values = values.str.lower()
1920 df[group] = values.to_numpy()
1921 return df
1924def source_file_regex_collisions(
1925 df: pd.DataFrame, pattern: str, *, column: str = SOURCE_FILE_COLUMN
1926) -> list:
1927 """Named groups of ``pattern`` that already exist as columns in ``df``.
1929 :func:`extract_columns_from_source_file` skips these (so it never clobbers
1930 real data); the wizard surfaces them so the user can rename the group.
1931 ``column`` is accepted for symmetry with its two siblings (UX-113 —
1932 "derive from any column", not only ``source_file``) but isn't otherwise
1933 used: a collision is a collision against ``df``'s columns regardless of
1934 which column the regex reads from."""
1935 if not pattern:
1936 return []
1937 try:
1938 groups = re.compile(pattern).groupindex
1939 except re.error:
1940 return []
1941 return [g for g in groups if g in df.columns]
1944def _renumber_word_ids_across_blocks(
1945 out: pd.DataFrame, *, screen_cols: list, block_col: str, word_col: str
1946) -> pd.DataFrame:
1947 """Reassign ``word_col`` sequentially per screen, in block-then-original
1948 order — the post-aggregation half of :func:`aggregate_char_boxes`'s block
1949 support (see its docstring). ``out`` is the already-aggregated frame (one
1950 row per word box), still carrying the ``_box_l``/``_box_t`` temp columns
1951 :func:`aggregate_char_boxes` computes just before calling this. A block's
1952 reading-order position is its own first box's top-then-left corner —
1953 geometry the aggregation already has, not a second field to ask for."""
1954 block_order = (
1955 out.groupby([*screen_cols, block_col], sort=False)[["_box_t", "_box_l"]]
1956 .first()
1957 .reset_index()
1958 .sort_values([*screen_cols, "_box_t", "_box_l"])
1959 )
1960 block_order["_block_rank"] = block_order.groupby(screen_cols, sort=False).cumcount()
1961 out = out.merge(
1962 block_order[[*screen_cols, block_col, "_block_rank"]],
1963 on=[*screen_cols, block_col],
1964 how="left",
1965 )
1966 out = out.sort_values([*screen_cols, "_block_rank", word_col], kind="stable")
1967 out[word_col] = out.groupby(screen_cols, sort=False).cumcount()
1968 return out.drop(columns="_block_rank").reset_index(drop=True)
1971def aggregate_char_boxes(
1972 df: pd.DataFrame, schema: dict[str, str | None]
1973) -> pd.DataFrame:
1974 """Collapse character-level AOI rows into one bounding box per word.
1976 For interest-area tables shipped one row per *character* (e.g. CJK corpora
1977 that have no whitespace word boundaries), aggregate the characters of each
1978 word — grouped by the mapped trial id + word id (plus participant / text id /
1979 screen id when mapped) — into a single bounding box: min/max over the mapped
1980 box columns, first value of every other column. A mapped screen id narrows
1981 the group the same way participant/text id do — a multi-screen (multipart)
1982 trial commonly restarts word numbering per screen, so without it a shared
1983 word id on two screens of the same trial would silently merge into one box.
1984 Run this on the RAW frame *before*
1985 :func:`normalize_words` (which expects one row per word box). ``schema`` is a
1986 word schema dict (field → source column). Returns ``df`` unchanged when the
1987 trial or word-id column isn't mapped, or no box columns are.
1989 UX-113 — ``schema["block"]``, when mapped, groups words into sub-screen
1990 blocks that each restart their own local word numbering (e.g. a
1991 comprehension question's stem/target/distractor answer blocks). Without
1992 it, two blocks' word 0 would silently aggregate into one merged box. When
1993 it *is* mapped, the word id is also renumbered sequentially per screen
1994 afterwards — in block-reading-order (each block's own top-then-left
1995 position, geometry rather than a second field the user would have to
1996 supply) then original word-id order — so ids stay unique within one
1997 screen, which matters because a mapped word id can be authoritative for
1998 fixation assignment (see ``measures.assign_fixations_to_words``)."""
1999 word_col = schema.get("word_id")
2000 trial = schema.get("trial")
2001 if not word_col or not trial:
2002 return df
2003 block_col = schema.get("block") or None
2004 group_cols = list(trial_mapping_columns(trial))
2005 # A mapped screen id must narrow the group like participant/text_id do —
2006 # a multi-screen trial (multipart) commonly restarts word numbering per
2007 # screen (e.g. MultiplEYE's reading pages), so without this a shared
2008 # word id on two screens of the same trial would silently merge into one
2009 # box, the same failure mode "block" fixes within a single screen.
2010 for key in ("participant", "text_id", "screen_id"):
2011 mapped = schema.get(key)
2012 if mapped:
2013 group_cols += trial_mapping_columns(mapped)
2014 if block_col:
2015 group_cols.append(block_col)
2016 group_cols.append(word_col)
2017 # De-dup, keep only columns actually present, and require the word id.
2018 group_cols = [c for c in dict.fromkeys(group_cols) if c in df.columns]
2019 if word_col not in group_cols:
2020 return df
2021 if block_col not in group_cols: # mapped, but not actually on this frame
2022 block_col = None
2024 has_xywh = all(schema.get(k) for k in ("x", "y", "width", "height"))
2025 has_edges = all(schema.get(k) for k in ("left", "right", "top", "bottom"))
2026 if not has_xywh and not has_edges:
2027 return df
2029 df = df.copy()
2030 if has_xywh:
2031 left = _to_number(df[schema["x"]])
2032 top = _to_number(df[schema["y"]])
2033 df["_box_l"], df["_box_t"] = left, top
2034 df["_box_r"] = left + _to_number(df[schema["width"]])
2035 df["_box_b"] = top + _to_number(df[schema["height"]])
2036 else:
2037 df["_box_l"] = _to_number(df[schema["left"]])
2038 df["_box_r"] = _to_number(df[schema["right"]])
2039 df["_box_t"] = _to_number(df[schema["top"]])
2040 df["_box_b"] = _to_number(df[schema["bottom"]])
2042 temp = {"_box_l", "_box_r", "_box_t", "_box_b"}
2043 agg = {c: "first" for c in df.columns if c not in group_cols and c not in temp}
2044 agg.update(_box_l="min", _box_t="min", _box_r="max", _box_b="max")
2045 out = df.groupby(group_cols, sort=False, as_index=False).agg(agg)
2047 if block_col:
2048 out = _renumber_word_ids_across_blocks(
2049 out,
2050 screen_cols=[c for c in group_cols if c not in (word_col, block_col)],
2051 block_col=block_col,
2052 word_col=word_col,
2053 )
2055 # Write the aggregated box back into the SAME schema columns so the existing
2056 # word schema still maps it (origin+size or edges, matching the input form).
2057 if has_xywh:
2058 out[schema["x"]] = out["_box_l"]
2059 out[schema["y"]] = out["_box_t"]
2060 out[schema["width"]] = out["_box_r"] - out["_box_l"]
2061 out[schema["height"]] = out["_box_b"] - out["_box_t"]
2062 else:
2063 out[schema["left"]] = out["_box_l"]
2064 out[schema["right"]] = out["_box_r"]
2065 out[schema["top"]] = out["_box_t"]
2066 out[schema["bottom"]] = out["_box_b"]
2067 return out.drop(columns=list(temp))
2070def _read_by_extension(
2071 buf, name: str, plan: ReadPlan | None = None, *, sep: str | None = None
2072) -> pd.DataFrame:
2073 """Dispatch a buffer/path to a pandas reader by its (lowercased) name.
2075 ``sep`` is the delimiter of a text table when the caller already knows it
2076 (a zip member, which cannot be peeked at); otherwise it is read off the
2077 header line (DATA-41).
2079 ``plan`` (PERF-6) narrows the read to the columns normalization keeps and
2080 declares EyeLink's ``.`` missing in the numeric ones. Delimited text and
2081 Parquet honour it; Excel has no column-projection reader, so it is read
2082 whole and pruned after.
2084 CSV/TSV reads pass ``low_memory=False`` so pandas infers one dtype per
2085 column in a single pass. The default chunked parser can otherwise read the
2086 same column as numeric in early chunks and as strings in a later chunk that
2087 holds a sentinel (e.g. EyeLink's ``.`` in ``CURRENT_FIX_PRECISION_MEASURE_*``
2088 columns), leaving a single ``object`` column that mixes Python ``float`` and
2089 ``str`` values. Such a column emits a ``DtypeWarning`` and later crashes
2090 pyarrow when Streamlit serializes the frame for display — only on large
2091 (multi-chunk) files, which is why a small upload reads fine locally but a
2092 full report kills the worker on the cloud. Matches the other read paths
2093 (``load_onestop_server_bundle``, ``onestop_shard``)."""
2094 columns = list(plan.columns) if plan is not None and plan.columns else None
2095 if name.endswith(".parquet"):
2096 return pd.read_parquet(buf, columns=columns)
2097 if name.endswith(".feather"):
2098 return pd.read_feather(buf, columns=columns)
2099 if name.endswith((".xlsx", ".xls")):
2100 if not _is_workbook(buf, name):
2101 # BUG-55: EyeLink Data Viewer's "Excel" export is tab-separated text
2102 # with an .xls name — read it as what it is.
2103 return _read_delimited(buf, sep or _sniff_delimiter(buf, name), plan)
2104 # First sheet (e.g. MultiplEYE questions workbook).
2105 frame = pd.read_excel(buf, **_excel_na_kwargs(buf, plan))
2106 return frame[[c for c in columns if c in frame.columns]] if columns else frame
2107 return _read_delimited(buf, sep or _sniff_delimiter(buf, name), plan)
2110#: The delimiters a text table is looked for with, and the one assumed when its
2111#: header line settles nothing (DATA-41).
2112_DELIMITERS = ("\t", ",", ";", "|")
2113_QUOTED = re.compile(r'"[^"]*"')
2116def _default_delimiter(name: str) -> str:
2117 """The delimiter a text file's extension implies: tab for ``.tsv`` /
2118 ``.tab`` and for the tab-separated exports named ``.txt`` or ``.xls``,
2119 comma for everything else."""
2120 return "\t" if name.lower().endswith((".tsv", ".tab", ".txt", ".xls")) else ","
2123def _delimiter_of(header_line: bytes, name: str) -> str:
2124 """The delimiter a table's header line uses (DATA-41).
2126 A ``;``-separated CSV (Excel's export wherever the decimal separator is a
2127 comma) and a tab-separated ``.txt`` both used to load as a single column
2128 holding the whole line. Counting each candidate in the header, outside
2129 quotes, is enough: a column name never contains the delimiter, while a
2130 sniffer that also reads the rows is misled by the decimal commas in them.
2131 A ``.tsv`` is always tab-separated; a header with no candidate in it (a
2132 one-column table) keeps the extension's default.
2133 """
2134 default = _default_delimiter(name)
2135 if name.lower().endswith((".tsv", ".tab")):
2136 return default
2137 text = _QUOTED.sub("", header_line.decode("latin-1"))
2138 counts = {sep: text.count(sep) for sep in _DELIMITERS}
2139 best = max(counts, key=lambda sep: counts[sep])
2140 return best if counts[best] > counts[default] else default
2143def _first_line(head: bytes) -> bytes:
2144 """The header line of a text table's opening bytes.
2146 A UTF-16 or UTF-32 file (EyeLink Data Viewer's default export) is decoded
2147 first, so the line is cut at its newline character rather than at the
2148 first newline *byte*, and comes back as UTF-8.
2149 """
2150 encoding = _bom_encoding(head)
2151 if encoding is not None and encoding != "utf-8-sig":
2152 line = head.decode(encoding, "ignore").split("\n", 1)[0].rstrip("\r")
2153 return line.encode("utf-8")
2154 return head.split(b"\n", 1)[0].rstrip(b"\r")
2157def _sniff_delimiter(buf, name: str) -> str:
2158 """:func:`_delimiter_of` for an upload or a path, read without consuming
2159 it; a stream that cannot be rewound keeps the extension's default."""
2160 if not _can_reread(buf):
2161 return _default_delimiter(name)
2162 return _delimiter_of(_first_line(_peek(buf, _HEADER_MAX_BYTES)), name)
2165#: The first bytes of a legacy (OLE2) Excel workbook, and of a zip container —
2166#: which is what an .xlsx is (BUG-55).
2167_OLE2_MAGIC = b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1"
2168_ZIP_MAGIC = b"PK\x03\x04"
2170#: Tried in order when a delimited file is not UTF-8 (BUG-55): Windows' default
2171#: for Western European text first, since that is what Excel writes there, then
2172#: Latin-1, which decodes any byte and so always ends the search.
2173_TEXT_ENCODINGS = ("utf-8", "cp1252", "latin-1")
2175#: Byte-order marks, longest first (UTF-32 LE's begins with UTF-16 LE's). A
2176#: file that opens with one is read in that encoding alone: the fallback above
2177#: would decode EyeLink Data Viewer's UTF-16 export as Latin-1, which never
2178#: fails, and every column but the first would come out as ``Unnamed: n``.
2179_BOMS = (
2180 (b"\xff\xfe\x00\x00", "utf-32"),
2181 (b"\x00\x00\xfe\xff", "utf-32"),
2182 (b"\xef\xbb\xbf", "utf-8-sig"),
2183 (b"\xff\xfe", "utf-16"),
2184 (b"\xfe\xff", "utf-16"),
2185)
2188def _bom_encoding(head: bytes) -> str | None:
2189 """The encoding a byte-order mark at the start of ``head`` declares."""
2190 for bom, encoding in _BOMS:
2191 if head.startswith(bom):
2192 return encoding
2193 return None
2196def _peek(file_like_or_path, size: int = 8) -> bytes:
2197 """The first ``size`` bytes of an upload or a path, leaving it rewound."""
2198 if hasattr(file_like_or_path, "read"):
2199 _rewind(file_like_or_path)
2200 head = file_like_or_path.read(size)
2201 _rewind(file_like_or_path)
2202 return head if isinstance(head, bytes) else str(head).encode()
2203 try:
2204 with open(file_like_or_path, "rb") as handle:
2205 return handle.read(size)
2206 except OSError:
2207 return b""
2210def _is_workbook(buf, name: str) -> bool:
2211 """Whether an Excel-named file is a real workbook pandas can open.
2213 A zip container is an .xlsx (openpyxl) and an OLE2 container is a legacy
2214 Excel 97–2003 workbook (xlrd, DATA-53); anything else is delimited text
2215 wearing an Excel extension (BUG-55).
2216 """
2217 head = _peek(buf)
2218 return head.startswith((_ZIP_MAGIC, _OLE2_MAGIC))
2221def _can_reread(buf) -> bool:
2222 """Whether ``buf`` can be read a second time from the start."""
2223 if isinstance(buf, (str, os.PathLike)):
2224 return True
2225 seekable = getattr(buf, "seekable", None)
2226 try:
2227 return bool(seekable()) if callable(seekable) else False
2228 except (OSError, ValueError):
2229 return False
2232def _read_delimited(buf, sep: str, plan: ReadPlan | None, **extra) -> pd.DataFrame:
2233 """``read_csv`` with the encoding fallback a non-UTF-8 export needs (BUG-55).
2235 A CSV saved by Excel on Windows is cp1252, and one umlaut in it made the
2236 whole upload fail with a raw ``UnicodeDecodeError``. Each encoding in
2237 ``_TEXT_ENCODINGS`` is tried in turn; a stream that cannot be rewound (a zip
2238 member) re-raises, and :func:`_read_zipped_table` retries it from memory.
2239 A byte-order mark settles the encoding outright (a UTF-16 Data Viewer
2240 export).
2241 """
2242 bom = _bom_encoding(_peek(buf, 4)) if _can_reread(buf) else None
2243 for encoding in (bom,) if bom else _TEXT_ENCODINGS:
2244 try:
2245 return pd.read_csv(
2246 buf,
2247 sep=sep,
2248 low_memory=False,
2249 encoding=encoding,
2250 **_read_kwargs(plan),
2251 **extra,
2252 )
2253 except UnicodeDecodeError:
2254 if bom or encoding == _TEXT_ENCODINGS[-1] or not _can_reread(buf):
2255 raise
2256 _rewind(buf)
2257 _LOGGER.info(
2258 "%s is not %s; reading it again as the next encoding",
2259 getattr(buf, "name", buf),
2260 encoding,
2261 )
2262 raise AssertionError("unreachable: latin-1 decodes every byte")
2265#: EyeLink writes a value it could not measure as a bare period. Declared
2266#: missing for *numeric* columns only: a word whose text is "." is a word
2267#: (PERF-6), so a blanket ``na_values="."`` would delete it from the stimulus.
2268MISSING_MARKER = "."
2270#: Schema fields whose values are numbers, and which therefore read the marker
2271#: above as missing. The identity fields (participant, trial, text_id, text,
2272#: screen_id) are deliberately absent — a "." there is a label.
2273NUMERIC_SCHEMA_FIELDS = frozenset(
2274 {
2275 "bottom",
2276 "canvas_height",
2277 "canvas_width",
2278 "duration",
2279 "fixation_id",
2280 "height",
2281 "left",
2282 "line",
2283 "right",
2284 "screen_index",
2285 "screen_timestamp",
2286 "timestamp",
2287 "top",
2288 "width",
2289 "word_id",
2290 "x",
2291 "y",
2292 }
2293)
2295#: One number written with a decimal comma (``117,7``) — how a German- or
2296#: French-locale export writes every fractional value (BUG-54).
2297_DECIMAL_COMMA = re.compile(r"^[+-]?\d+,\d+$")
2298#: ...and the shape a *thousands* separator gives the same characters
2299#: (``1,204``). A column whose every comma looks like this could be either, so it
2300#: is reported rather than guessed at.
2301_THOUSANDS_GROUPED = re.compile(r"^[+-]?[1-9]\d{0,2}(,\d{3})+$")
2304def _filled_cells(values: pd.Series) -> pd.Series:
2305 """The cells of ``values`` that hold something, as stripped text.
2307 EyeLink's ``.`` marker and a blank cell are *missing*, not unreadable, so
2308 they are left out — a planned read already turned them into NaN, and an
2309 unplanned one must not report them as garbage either.
2310 """
2311 text = values[values.notna()].astype(str).str.strip()
2312 return text[(text != "") & (text != MISSING_MARKER)]
2315def _unparsed_cells(values: pd.Series, parsed: pd.Series) -> pd.Series:
2316 """The filled cells of ``values`` that ``parsed`` could not read, as text."""
2317 failed = parsed.isna() & values.notna()
2318 if not failed.any():
2319 return pd.Series([], dtype=str)
2320 return _filled_cells(values[failed])
2323def _to_number(values: pd.Series) -> pd.Series:
2324 """``pd.to_numeric(errors="coerce")``, reading a decimal-comma column too.
2326 A decimal-comma export (``117,7``) made every fractional cell unparseable,
2327 and NaN then fell through to the silent fallbacks downstream — each fixation
2328 snapped to its word's centre, each duration read as 0 (BUG-54). The commas
2329 are converted only when **every** cell that failed is a decimal-comma number
2330 and not all of them could be a thousands separator instead; anything else
2331 stays NaN, for :func:`numeric_parse_issues` to report.
2332 """
2333 parsed = pd.to_numeric(values, errors="coerce")
2334 if pd.api.types.is_numeric_dtype(values):
2335 return parsed
2336 failed = _unparsed_cells(values, parsed)
2337 if failed.empty or not failed.str.fullmatch(_DECIMAL_COMMA).all():
2338 return parsed
2339 if failed.str.fullmatch(_THOUSANDS_GROUPED).all():
2340 return parsed
2341 converted = pd.to_numeric(failed.str.replace(",", ".", regex=False))
2342 parsed = parsed.copy()
2343 parsed.loc[converted.index] = converted
2344 return parsed
2347#: How a mapped field's values are read, for :func:`mapping_value_preview`.
2348_PREVIEW_ID_FIELDS = frozenset(
2349 {
2350 "participant",
2351 "trial",
2352 "text_id",
2353 "word_id",
2354 "fixation_id",
2355 "screen_id",
2356 "screen_fixation_id",
2357 "block",
2358 }
2359)
2360_PREVIEW_TIME_FIELDS = frozenset({"duration", "timestamp", "screen_timestamp"})
2361_PREVIEW_PIXEL_FIELDS = frozenset(
2362 {
2363 "x",
2364 "y",
2365 "width",
2366 "height",
2367 "left",
2368 "right",
2369 "top",
2370 "bottom",
2371 "canvas_width",
2372 "canvas_height",
2373 }
2374)
2375_PREVIEW_NUMBER_FIELDS = frozenset({"line", "screen_index"}) | {
2376 key for key, *_ in READING_MEASURE_FIELDS
2377}
2378#: Rows looked at — the first few values are all a preview shows, and reading
2379#: the head keeps it free on a table of millions of rows.
2380_PREVIEW_ROWS = 200
2383def mapping_value_preview(
2384 df: pd.DataFrame | None, field_key: str, column, *, limit: int = 3
2385) -> str:
2386 """A few of ``column``'s values and what the app reads them as.
2388 The mapping editor's value preview: a plausible column name can still hold
2389 the wrong thing — trial ids picked as a condition, an onset as a duration,
2390 seconds read as milliseconds — and its first values show it before saving.
2391 Reads the head of the frame the editor already holds, through the same
2392 conversions normalization applies (:func:`stable_id`, :func:`_to_number`,
2393 :func:`time_unit_ms`). ``""`` when there is nothing to show.
2394 """
2395 if df is None or not column:
2396 return ""
2397 columns = [str(c) for c in trial_mapping_columns(column)]
2398 if not columns or any(c not in df.columns for c in columns):
2399 return ""
2400 head = df.head(_PREVIEW_ROWS)[columns].dropna(how="all")
2401 if head.empty:
2402 return "No values in the first rows"
2404 def number(value: float) -> str:
2405 return f"{value:,.6g}"
2407 if len(columns) > 1 or field_key in _PREVIEW_ID_FIELDS:
2408 ids = trial_id_series(head, column).drop_duplicates().head(limit)
2409 shown = []
2410 for index, value in ids.items():
2411 source = " + ".join(str(head.at[index, c]) for c in columns)
2412 shown.append(value if source == value else f"{source} → {value}")
2413 return "Read as IDs: " + ", ".join(shown)
2414 values = head[columns[0]].dropna().head(limit)
2415 if (
2416 field_key in _PREVIEW_TIME_FIELDS
2417 or field_key in _PREVIEW_PIXEL_FIELDS
2418 or field_key in _PREVIEW_NUMBER_FIELDS
2419 ):
2420 parsed = _to_number(values)
2421 if parsed.isna().all():
2422 return "Not numbers: " + ", ".join(str(v) for v in values)
2423 factor = 1.0
2424 unit = ""
2425 if field_key in _PREVIEW_TIME_FIELDS:
2426 factor, unit = time_unit_ms(columns[0]), " ms"
2427 elif field_key in _PREVIEW_PIXEL_FIELDS:
2428 unit = " px"
2429 shown = []
2430 for source, value in zip(values, parsed):
2431 if pd.isna(value):
2432 shown.append(f"{source} (not a number)")
2433 elif factor != 1.0:
2434 shown.append(f"{source} → {number(value * factor)}{unit}")
2435 else:
2436 shown.append(f"{number(value)}{unit}")
2437 return ", ".join(shown)
2438 return ", ".join(f"“{value}”" for value in values.astype(str))
2441#: What becomes of a fixation or word whose mapped numeric cell is unreadable —
2442#: said in the warning, because "left empty" means something different per field.
2443_UNPARSED_CONSEQUENCE = {
2444 "duration": "those fixations are read as 0 ms long",
2445 "timestamp": "those fixations are read as starting at 0",
2446 "x": "those fixations are placed at their word's center when they have a word id, "
2447 "and left off the plot otherwise",
2448 "y": "those fixations are placed at their word's center when they have a word id, "
2449 "and left off the plot otherwise",
2450 "word_id": "those rows have no word id",
2451}
2452#: The same for the Words table, where x/y are a box's corner, not a gaze (#374).
2453_UNPARSED_WORD_CONSEQUENCE = dict.fromkeys(
2454 ("x", "y", "width", "height", "left", "right", "top", "bottom"),
2455 "those word boxes have no position",
2456)
2459def numeric_parse_issues(
2460 raw: pd.DataFrame, schema: dict, *, table: str, fixations: bool = True
2461) -> list[str]:
2462 """Plain-language warnings for mapped numeric columns that did not parse.
2464 One line per column, naming the table, the column, how many of its cells
2465 were unreadable, a few examples, and what the load did with those rows —
2466 because the load carries on either way, and a column read as all-NaN used to
2467 produce a plausible-looking figure with nothing said (BUG-54). A
2468 decimal-comma column that :func:`_to_number` converts is not an issue; one
2469 whose commas could equally be thousands separators is, with that named.
2470 """
2471 issues: list[str] = []
2472 seen: set = set()
2473 for key, column in schema.items():
2474 if key not in NUMERIC_SCHEMA_FIELDS or not isinstance(column, str):
2475 continue
2476 if column in seen or column not in raw.columns:
2477 continue
2478 seen.add(column)
2479 values = raw[column]
2480 if pd.api.types.is_numeric_dtype(values) or pd.api.types.is_bool_dtype(values):
2481 continue
2482 failed = _unparsed_cells(values, _to_number(values))
2483 if failed.empty:
2484 continue
2485 examples = ", ".join(f"'{v}'" for v in failed.drop_duplicates().head(3))
2486 line = (
2487 f"{table}: {len(failed):,} of {len(_filled_cells(values)):,} values in "
2488 f"`{column}` aren't numbers (e.g. {examples})"
2489 )
2490 if failed.str.fullmatch(_THOUSANDS_GROUPED).all():
2491 line += (
2492 ". They could be a decimal comma or a thousands separator, so they "
2493 "were not guessed at — re-export the table with a '.' decimal point "
2494 "and no thousands separator"
2495 )
2496 consequences = (
2497 _UNPARSED_CONSEQUENCE if fixations else _UNPARSED_WORD_CONSEQUENCE
2498 )
2499 consequence = consequences.get(key, "those cells are left empty")
2500 issues.append(f"{line}; {consequence}.")
2501 return issues
2504def _identity_columns(source: pd.DataFrame, schema: dict) -> list[str]:
2505 """The source columns a row's (participant, trial) identity is built from."""
2506 columns: list = []
2507 if schema.get("participant"):
2508 columns += trial_mapping_columns(schema["participant"])
2509 columns += trial_mapping_columns(schema["trial"])
2510 return [c for c in dict.fromkeys(columns) if c in source.columns]
2513def _id_missing(values: pd.Series) -> pd.Series:
2514 """Cells that hold no id: missing, or text that is only whitespace — which
2515 :func:`stable_id` would turn into an empty id, a trial named ``""``."""
2516 missing = values.isna()
2517 if pd.api.types.is_numeric_dtype(values) or pd.api.types.is_bool_dtype(values):
2518 return missing
2519 # Tested per distinct value: an id column has few, and a corpus many rows.
2520 distinct = pd.Series(values[~missing].unique())
2521 blank = distinct[distinct.astype(str).str.strip().eq("")]
2522 return missing | values.isin(blank) if len(blank) else missing
2525def _rows_missing_identity(
2526 source: pd.DataFrame, schema: dict
2527) -> tuple[pd.Series, pd.Series]:
2528 """``(blank, unkeyed)`` row masks: rows with no participant or trial id.
2530 A row missing either cannot belong to any trial, and its NaN crashed the
2531 load outright — one ``,,,,`` line, the blank row Excel leaves at the end of
2532 a sheet, made the whole dataset impossible to add (BUG-56). ``blank`` is the
2533 rows that hold nothing at all, which are not data and go quietly;
2534 ``unkeyed`` is the rest, which hold data and are reported. An id that is
2535 only whitespace counts as missing (round 10). Only the missing rows are
2536 inspected cell by cell, so a clean table costs one ``isna`` and one pass
2537 over the distinct values per id column.
2538 """
2539 columns = _identity_columns(source, schema)
2540 missing = source[columns].apply(_id_missing).any(axis=1) if columns else None
2541 none = pd.Series(False, index=source.index)
2542 if missing is None or not missing.any():
2543 return none, none
2544 rows = source.loc[missing].drop(columns=[SOURCE_FILE_COLUMN], errors="ignore")
2545 empty = rows.apply(lambda c: c.isna() | (c.astype(str).str.strip() == ""))
2546 blank = none.copy()
2547 blank.loc[rows.index] = empty.all(axis=1)
2548 return blank, missing & ~blank
2551def _drop_rows_missing_identity(source: pd.DataFrame, schema: dict) -> pd.DataFrame:
2552 """``source`` without the rows :func:`_rows_missing_identity` flags."""
2553 blank, unkeyed = _rows_missing_identity(source, schema)
2554 keep = ~(blank | unkeyed)
2555 return source if keep.all() else source.loc[keep]
2558def identity_issues(
2559 raw: pd.DataFrame,
2560 schema: dict,
2561 *,
2562 table: str,
2563 unkeyed: pd.Series | None = None,
2564) -> list[str]:
2565 """A warning for rows that hold data but no participant or trial id.
2567 ``unkeyed`` is :func:`_rows_missing_identity`'s second mask, when the
2568 caller already has it."""
2569 if unkeyed is None:
2570 _, unkeyed = _rows_missing_identity(raw, schema)
2571 count = int(unkeyed.sum())
2572 if not count:
2573 return []
2574 columns = [
2575 c
2576 for c in _identity_columns(raw, schema)
2577 if _id_missing(raw.loc[unkeyed, c]).any()
2578 ]
2579 named = ", ".join(f"`{c}`" for c in columns)
2580 if count == 1:
2581 said = "1 row has no value in {}, so it belongs to no trial and was left out"
2582 else:
2583 said = f"{count:,} rows have no value in {{}}, so they belong to no trial and were left out"
2584 return [f"{table}: {said.format(named)}."]
2587#: Positions that never leave this band are fractions of the screen, not
2588#: pixels: Gazepoint's FPOGX/FPOGY and Pupil Labs Core's norm_pos_x/y. Wider
2589#: than 0–1, because a fraction strays a little off-screen.
2590_FRACTION_BAND = (-0.5, 1.5)
2593def screen_fraction_issues(raw: pd.DataFrame, schema: dict, *, table: str) -> list:
2594 """A warning when the mapped X/Y are screen fractions rather than pixels.
2596 Not converted (DATA-40): the load knows no screen size to scale by, and a
2597 guessed one would put every fixation in the wrong place while looking
2598 plausible — the one outcome worse than a figure that is obviously wrong.
2599 """
2600 x, y = schema.get("x"), schema.get("y")
2601 if not (isinstance(x, str) and isinstance(y, str)):
2602 return []
2603 if x not in raw.columns or y not in raw.columns:
2604 return []
2605 xs, ys = _to_number(raw[x]), _to_number(raw[y])
2606 both = xs.notna() & ys.notna()
2607 if not both.any():
2608 return []
2609 low, high = _FRACTION_BAND
2610 xs, ys = xs[both], ys[both]
2611 if not (xs.between(low, high).all() and ys.between(low, high).all()):
2612 return []
2613 if not (xs.between(0, 1, inclusive="neither").any()):
2614 return [] # all 0 / all 1 is degenerate data, not a fraction
2615 return [
2616 f"{table}: every position in `{x}` / `{y}` lies between 0 and 1 — these "
2617 "look like fractions of the screen (Gazepoint's FPOGX/FPOGY, Pupil Labs "
2618 "Core's norm_pos), not pixels, so the scanpath is drawn in a 1-pixel "
2619 "corner of the canvas. They are not converted, because the screen size "
2620 "is not known here: multiply them by the screen width and height in "
2621 "pixels before uploading (for Pupil Core, whose y points up, use "
2622 "(1 − y) × height)."
2623 ]
2626def normalization_issues(
2627 raw: pd.DataFrame, schema: dict, *, table: str, fixations: bool = False
2628) -> list[str]:
2629 """Everything the load will do to ``raw`` under ``schema`` that the user
2630 should hear about: rows left out for want of an id (BUG-56), mapped numeric
2631 columns that did not parse (BUG-54), and — for ``fixations``, whose X/Y
2632 are gaze positions rather than box origins — positions that are screen
2633 fractions rather than pixels (DATA-40)."""
2634 issues = identity_issues(raw, schema, table=table)
2635 issues += numeric_parse_issues(raw, schema, table=table, fixations=fixations)
2636 if fixations:
2637 issues += screen_fraction_issues(raw, schema, table=table)
2638 return issues
2641def _warn_normalization_issues(
2642 raw: pd.DataFrame, schema: dict, *, table: str, fixations: bool = False
2643) -> None:
2644 """Raise each :func:`normalization_issues` line as a ``UserWarning``.
2646 The headless API and ``render`` have no page to put a warning on, so the
2647 normalizers say it themselves; the wizard shows the same lines above
2648 ✅ Add dataset.
2649 """
2650 for issue in normalization_issues(raw, schema, table=table, fixations=fixations):
2651 warnings.warn(issue.replace("`", "'"), UserWarning, stacklevel=3)
2654@dataclass(frozen=True)
2655class ReadPlan:
2656 """Which columns to parse out of a table, and what counts as missing in them.
2658 ``columns`` is ``None`` for "parse everything" — the honest answer when the
2659 mapping claims nothing in the header, so there is no basis on which to drop
2660 anything.
2661 """
2663 columns: tuple[str, ...] | None = None
2664 na_values: dict[str, list[str]] = field(default_factory=dict)
2665 #: Columns read as the literal text of each cell, with no cell taken as
2666 #: missing (BUG-53). The word-text column: "None", "NA" and "null" are
2667 #: words a stimulus can contain, and pandas' default NA spellings turned
2668 #: every one of them into NaN before normalization saw it.
2669 verbatim: tuple[str, ...] = ()
2670 #: Identity columns (participant, trial, text, screen) read as text, so a
2671 #: zero-padded id survives: CSV inference read `007` as the number 7, while
2672 #: the same id in a Parquet table stayed "007", and the two tables then
2673 #: shared no participant at all (BUG-59). Missing cells stay missing.
2674 identity: tuple[str, ...] = ()
2676 def narrowed_to(self, available: Iterable[str]) -> ReadPlan:
2677 """This plan restricted to the columns one file actually has.
2679 A multi-file read — several uploads, or several members of one zip —
2680 plans from the *union* of their headers, and ``usecols`` raises on a
2681 name the file it is reading does not carry. Narrowing per file is what
2682 keeps a heterogeneous set readable, exactly as it was before the read
2683 was planned at all (``read_tables``: "fields absent from a file become
2684 NaN"). An empty intersection degrades to reading the whole file rather
2685 than to reading none of it: a file this plan cannot describe is one
2686 there is no basis to prune.
2687 """
2688 if self.columns is None:
2689 return self
2690 present = set(available)
2691 columns = tuple(name for name in self.columns if name in present)
2692 verbatim = tuple(c for c in self.verbatim if c in present)
2693 identity = tuple(c for c in self.identity if c in present)
2694 if not columns:
2695 return ReadPlan(verbatim=verbatim, identity=identity)
2696 return ReadPlan(
2697 columns=columns,
2698 na_values={k: v for k, v in self.na_values.items() if k in present},
2699 verbatim=verbatim,
2700 identity=identity,
2701 )
2704def _read_kwargs(plan: ReadPlan | None) -> dict:
2705 """``read_csv`` keywords for a plan (nothing at all for ``None``).
2707 An empty ``columns`` reads the whole file, matching the columnar readers in
2708 :func:`_read_by_extension` — `usecols=[]` would parse nothing at all, and
2709 the two must not disagree about what an empty plan means.
2710 """
2711 if plan is None:
2712 return {}
2713 kwargs: dict = {}
2714 if plan.columns:
2715 kwargs["usecols"] = list(plan.columns)
2716 if plan.na_values:
2717 kwargs["na_values"] = plan.na_values
2718 if plan.verbatim:
2719 # A converter receives the cell's raw text before NA detection runs, and
2720 # leaves every other column's NA handling exactly as it was (BUG-53).
2721 kwargs["converters"] = {column: str for column in plan.verbatim}
2722 if plan.identity:
2723 kwargs["dtype"] = {column: str for column in plan.identity}
2724 return kwargs
2727#: pandas' own default missing-value spellings (``read_csv``'s ``na_values``
2728#: docs). Spelled out because an Excel read with a verbatim column has to switch
2729#: the defaults off and hand them back to every *other* column by name.
2730PANDAS_DEFAULT_NA = frozenset(
2731 {
2732 "",
2733 "#N/A",
2734 "#N/A N/A",
2735 "#NA",
2736 "-1.#IND",
2737 "-1.#QNAN",
2738 "-NaN",
2739 "-nan",
2740 "1.#IND",
2741 "1.#QNAN",
2742 "<NA>",
2743 "N/A",
2744 "NA",
2745 "NULL",
2746 "NaN",
2747 "None",
2748 "n/a",
2749 "nan",
2750 "null",
2751 }
2752)
2755def _excel_na_kwargs(buf, plan: ReadPlan | None) -> dict:
2756 """``read_excel`` keywords that keep a plan's verbatim columns literal.
2758 ``read_excel`` applies its NA spellings before a converter sees the cell,
2759 so the CSV path's converter trick does not reach it: the defaults are turned
2760 off and given back to every other column by name, which needs the header
2761 first. Excel is never the large-file format, so the second pass is cheap.
2762 """
2763 if plan is None:
2764 return {}
2765 kwargs: dict = {}
2766 if plan.identity:
2767 kwargs["dtype"] = {column: str for column in plan.identity}
2768 if not plan.verbatim:
2769 return kwargs
2770 header = list(pd.read_excel(buf, nrows=0).columns)
2771 _rewind(buf)
2772 na_values = {
2773 column: sorted(PANDAS_DEFAULT_NA | set(plan.na_values.get(column, ())))
2774 for column in header
2775 if column not in plan.verbatim
2776 }
2777 return {**kwargs, "keep_default_na": False, "na_values": na_values}
2780def verbatim_text_plan(header: Sequence[str], schema: dict | None = None) -> ReadPlan:
2781 """A whole-table words plan: the word text verbatim, the ids as text.
2783 For readers that parse every column (the headless API) but still must not
2784 lose a word spelled "None" or "NA" (BUG-53), nor merge reader ``01`` into
2785 reader ``1`` by reading the ids as numbers. ``schema`` is the caller's own
2786 word mapping; without one the columns are auto-detected from the header,
2787 the way the mapping itself will be.
2788 """
2789 names = list(header)
2790 schema = schema or propose_word_schema(pd.DataFrame(columns=names))
2791 text = schema.get("text")
2792 verbatim = (text,) if isinstance(text, str) and text in names else ()
2793 return ReadPlan(
2794 verbatim=verbatim, identity=_identity_columns_in(names, schema, verbatim)
2795 )
2798def identity_text_plan(
2799 header: Sequence[str], schema: dict | None = None, *, kind: str = "fixations"
2800) -> ReadPlan:
2801 """A whole-table plan that reads only the identity columns as text.
2803 The headless counterpart of what :func:`plan_table_read` does for the app,
2804 for the tables whose word text is not at stake (fixations, raw gaze): the
2805 participant / trial / text / screen columns — the caller's ``schema``'s,
2806 else auto-detected from the header, composite ids expanded — are read as
2807 text, so ``01`` and ``1`` stay two readers. Every column is still parsed.
2808 """
2809 proposers = {
2810 "fixations": propose_fix_schema,
2811 "raw_gaze": propose_raw_gaze_schema,
2812 }
2813 if kind not in proposers:
2814 raise ValueError(f"kind must be one of {sorted(proposers)}, not {kind!r}")
2815 names = list(header)
2816 schema = schema or proposers[kind](pd.DataFrame(columns=names))
2817 return ReadPlan(identity=_identity_columns_in(names, schema, ()))
2820def _identity_columns_in(
2821 names: Sequence[str], schema: dict, verbatim: Sequence[str]
2822) -> tuple[str, ...]:
2823 """The schema's identity source columns that this header carries."""
2824 present = set(names)
2825 return tuple(
2826 column
2827 for column in dict.fromkeys(
2828 [*_schema_identity_columns(schema), *_IDENTITY_SOURCES]
2829 )
2830 if column in present and column not in verbatim and column not in _ORDINALS
2831 )
2834def plan_table_read(
2835 header: Sequence[str],
2836 schema: dict,
2837 registry: Sequence[tuple],
2838 *,
2839 filter_fields: Iterable[str] | None = None,
2840 keep_columns: Iterable[str] | None = None,
2841 text_column: str | None = None,
2842 identity_columns: Iterable[str] = (),
2843) -> ReadPlan:
2844 """Narrow a read to the columns ``normalize_*`` keeps (PERF-6).
2846 An EyeLink IA report ships 174 columns and a fixation report 300; the
2847 mapping plus ``*_OPTIONAL_FIELDS`` claim 27 and 17, and normalization drops
2848 the rest on the next line. Parsing them anyway is the largest single cost on
2849 the load path — measured on the full OneStop reports, planning the read
2850 takes it from ~200 s / 25 GB to ~58 s / 2.2 GB for a byte-identical frame.
2852 ``header`` is the column names alone (see :func:`read_table_columns`), so
2853 the plan is made before a single row is parsed. ``filter_fields`` and
2854 ``keep_columns`` carry the columns the user chose to keep beyond the
2855 mapping, exactly as :func:`compute_keep_columns` takes them.
2857 The word-text column — ``text_column`` when the user has mapped one by
2858 hand, else the schema's own ``text`` — is read verbatim (BUG-53), and the
2859 identity columns — the schema's, plus any ``identity_columns`` the user
2860 picked by hand — as text (BUG-59).
2861 """
2862 names = list(header)
2863 present = set(names)
2864 text = text_column or schema.get("text")
2865 verbatim = (text,) if isinstance(text, str) and text in present else ()
2866 identity = tuple(
2867 column
2868 for column in dict.fromkeys(
2869 [*_schema_identity_columns(schema), *identity_columns, *_IDENTITY_SOURCES]
2870 )
2871 if column in present and column not in verbatim and column not in _ORDINALS
2872 )
2873 if not (set(_schema_source_columns(schema)) & present):
2874 # Nothing is mapped yet — an unmapped upload, or a table this schema
2875 # does not describe. Dropping columns here would be guessing.
2876 return ReadPlan(verbatim=verbatim, identity=identity)
2877 keep = compute_keep_columns(
2878 schema,
2879 optional_sources=[row[0] for row in registry if row[0] in present],
2880 filter_fields=filter_fields,
2881 keep_columns=set(keep_columns or ()) | set(verbatim),
2882 )
2883 numeric = {row[0] for row in registry if row[2] == "numeric"}
2884 numeric |= {
2885 column
2886 for key, column in schema.items()
2887 if key in NUMERIC_SCHEMA_FIELDS and isinstance(column, str)
2888 }
2889 columns = tuple(name for name in names if name in keep)
2890 return ReadPlan(
2891 columns=columns,
2892 na_values={name: [MISSING_MARKER] for name in columns if name in numeric},
2893 verbatim=verbatim,
2894 identity=tuple(c for c in identity if c in keep),
2895 )
2898#: Schema fields that name *which* participant / trial / text / screen a row
2899#: belongs to — read as text, never as numbers (BUG-59).
2900IDENTITY_SCHEMA_FIELDS = ("participant", "trial", "text_id", "screen_id")
2901#: The id columns `normalize_*` consults by name rather than through the schema.
2902_IDENTITY_SOURCES = ("unique_trial_id", "unique_paragraph_id")
2903#: ...except an index that is also carried as a number: the trial picker sorts
2904#: on `TRIAL_INDEX`, and as text 10 would sort before 2.
2905_ORDINALS = frozenset({"TRIAL_INDEX", "trial_index"})
2908def _schema_identity_columns(schema: dict) -> list[str]:
2909 """The source columns a schema's identity fields name (lists expanded)."""
2910 columns: list = []
2911 for key in IDENTITY_SCHEMA_FIELDS:
2912 value = schema.get(key)
2913 if value:
2914 columns += trial_mapping_columns(value)
2915 return columns
2918def read_table_columns(file_like_or_path, *, kind: str | None = None) -> list[str]:
2919 """A table's column names, without parsing its rows.
2921 The header pass that :func:`plan_table_read` plans from. Delimited text
2922 reads zero rows and columnar formats read their schema; anything else
2923 (Excel, a zip of several members) falls back to reading the table, because
2924 there is no cheaper way to learn its columns and those formats are not the
2925 ones that hurt. ``kind`` (``"words"`` / ``"fixations"``) picks a mixed
2926 zip's members the way :func:`zip_member_split` does (#374, F3).
2927 """
2928 name = getattr(file_like_or_path, "name", str(file_like_or_path)).lower()
2929 _rewind(file_like_or_path)
2930 try:
2931 if name.endswith(".zip"):
2932 return _zipped_table_columns(file_like_or_path, kind=kind)
2933 if name.endswith(".parquet"):
2934 import pyarrow.parquet as pq
2936 return list(pq.read_schema(file_like_or_path).names)
2937 if name.endswith(".feather"):
2938 from pyarrow import feather
2940 return list(feather.read_table(file_like_or_path, columns=[]).schema.names)
2941 if name.endswith((".tsv", ".tab", ".csv", ".txt")):
2942 return _header(file_like_or_path, _sniff_delimiter(file_like_or_path, name))
2943 # Rewound afterwards too (the `finally`): the caller reads the table
2944 # again, and a buffer left at its end reads as an empty file.
2945 return list(read_table(file_like_or_path).columns)
2946 except pd.errors.EmptyDataError as exc:
2947 raise _empty_file_error(name) from exc
2948 finally:
2949 _rewind(file_like_or_path)
2952def _header(buf, sep: str) -> list[str]:
2953 """A delimited table's column names, read under the encoding fallback."""
2954 return list(_read_delimited(buf, sep, None, nrows=0).columns)
2957def _empty_file_error(name: str) -> ValueError:
2958 """The error an empty upload raises, in place of pandas' "No columns to
2959 parse from file" (BUG-55)."""
2960 return ValueError(
2961 f"'{Path(name).name}' is empty — it has no header row. Check the export "
2962 "finished writing, then upload it again."
2963 )
2966#: How far into a zip member the header line is looked for.
2967_HEADER_MAX_BYTES = 1024 * 1024
2970def _member_layout(zf: zipfile.ZipFile, info: zipfile.ZipInfo) -> tuple[list, str]:
2971 """Column names and delimiter of one delimited zip member, from its first
2972 line alone.
2974 A member stream cannot be rewound, so the encoding fallback runs over just
2975 the header line held in memory — cut at the newline, never mid-character —
2976 and the delimiter (DATA-41) is read off the same line.
2977 """
2978 with zf.open(info) as inner:
2979 head = inner.read(_HEADER_MAX_BYTES)
2980 line = _first_line(head)
2981 if not line.strip():
2982 raise _empty_file_error(info.filename)
2983 sep = _delimiter_of(line, info.filename)
2984 return _header(io.BytesIO(line + b"\n"), sep), sep
2987def _zipped_table_columns(file_like_or_path, *, kind: str | None = None) -> list[str]:
2988 """Column names of a ``.zip``'s members, without decompressing the rows.
2990 A OneStop report is a single ~4 GB CSV inside its zip, so the fallback of
2991 reading the table and taking its columns would unpack the whole archive to
2992 answer a question the first line already does. Members are unioned in order,
2993 matching how :func:`_read_zipped_table` concatenates them.
2994 """
2995 columns: list[str] = []
2996 with zipfile.ZipFile(file_like_or_path) as zf:
2997 infos = [
2998 i
2999 for i in zf.infolist()
3000 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__"))
3001 ]
3002 if not infos:
3003 raise ValueError(
3004 "the zip holds no table file (CSV, TSV, TXT, Excel, Parquet or Feather)"
3005 )
3006 # DATA-16/S6: the same declared-size guard the full read applies. A
3007 # cheaper way to learn an archive's columns must not also be a way
3008 # around its decompression limits.
3009 _check_zip_limits(infos)
3010 infos = _zip_split(zf, infos, kind).used_infos
3011 for info in infos:
3012 name = info.filename.lower()
3013 if not name.endswith((".tsv", ".tab", ".csv", ".txt")):
3014 # Columnar/workbook members seek, so there is no header-only
3015 # read: defer to `_read_zipped_table`, which reads every member
3016 # under a running byte budget (declared sizes are forgeable, so
3017 # the check above is not sufficient on its own).
3018 return list(_read_zipped_table(file_like_or_path, kind=kind).columns)
3019 names, _sep = _member_layout(zf, info)
3020 columns.extend(c for c in names if c not in columns)
3021 return columns
3024def _rewind(file_like_or_path) -> None:
3025 """Seek an uploaded buffer back to the start, so it can be read again."""
3026 seek = getattr(file_like_or_path, "seek", None)
3027 if callable(seek):
3028 seek(0)
3031def read_mapped_table(
3032 file_like_or_path,
3033 *,
3034 kind: str,
3035 declared: dict | None = None,
3036 filter_fields: Iterable[str] | None = None,
3037) -> pd.DataFrame:
3038 """Read one words or fixations table, parsing only the columns it needs.
3040 The header-then-plan-then-read pass in one call — what a loader that already
3041 knows which table it is reading wants (PERF-6). ``declared`` is a publisher's
3042 own mapping, used in place of auto-detection; ``filter_fields`` are extra
3043 source columns to keep for trial filtering.
3045 The frame comes back *pre-normalization*, exactly as a plain
3046 :func:`read_table` would return it, minus the columns nothing claims.
3047 """
3048 proposers = {
3049 "words": (propose_word_schema, WORD_OPTIONAL_FIELDS),
3050 "fixations": (propose_fix_schema, FIX_OPTIONAL_FIELDS),
3051 }
3052 if kind not in proposers:
3053 raise ValueError(f"kind must be one of {sorted(proposers)}, not {kind!r}")
3054 propose, registry = proposers[kind]
3055 header = read_table_columns(file_like_or_path)
3056 schema = dict(declared) if declared else propose(pd.DataFrame(columns=header))
3057 plan = plan_table_read(header, schema, registry, filter_fields=filter_fields)
3058 return read_table(file_like_or_path, plan=plan)
3061def source_labels(paths: Sequence[str]) -> list[str]:
3062 """The ``source_file`` label for each path — its stem, unless that stem
3063 is shared with another path.
3065 ``reader-a/fixations.csv`` and ``reader-b/fixations.csv`` both have the
3066 stem ``fixations``, and a label is mapped as participant or trial identity,
3067 so two readers would silently become one. A shared stem is qualified by
3068 the fewest trailing folders that tell it apart from the paths it clashes
3069 with (``reader-a/fixations``) — only folders that *differ*, so the shared
3070 part of an absolute path (``/Users/<name>/…``) never enters a label. Paths
3071 in the same folder then keep their extension (``fix.csv`` / ``fix.tsv``),
3072 and paths that are the same get a ``#n`` occurrence number. Backslashes
3073 count as folder separators, so a zip made on Windows labels as one made
3074 elsewhere.
3076 A browser upload carries no folders, so two same-named uploads read
3077 ``fixations#1`` / ``fixations#2`` in the app where the same files read from
3078 disk (API, CLI) get their folders. :func:`source_file_name` recovers the
3079 file name from any of these forms.
3080 """
3081 split = []
3082 for path in paths:
3083 parts = [
3084 p for p in str(path).replace("\\", "/").split("/") if p not in ("", ".")
3085 ]
3086 parts = parts or [str(path)]
3087 split.append((tuple(parts[:-1]), parts[-1]))
3088 stems = [Path(name).stem for _, name in split]
3090 def shared_tail(a: tuple, b: tuple) -> int:
3091 n = 0
3092 while n < min(len(a), len(b)) and a[-1 - n] == b[-1 - n]:
3093 n += 1
3094 return n
3096 labels = list(stems)
3097 for i, (folders, name) in enumerate(split):
3098 clashes = [j for j, stem in enumerate(stems) if j != i and stem == stems[i]]
3099 if not clashes:
3100 continue
3101 others = [split[j][0] for j in clashes if split[j][0] != folders]
3102 depth = min(
3103 len(folders), max((shared_tail(folders, o) + 1 for o in others), default=0)
3104 )
3105 same_folder = any(split[j][0] == folders for j in clashes)
3106 tail = Path(name).name if same_folder else stems[i]
3107 if same_folder and any(split[j] == split[i] for j in clashes):
3108 tail = stems[i]
3109 labels[i] = "/".join([*folders[len(folders) - depth :], tail])
3110 seen: dict[str, int] = {}
3111 counts: dict[str, int] = {}
3112 for label in labels:
3113 counts[label] = counts.get(label, 0) + 1
3114 for i, label in enumerate(labels):
3115 if counts[label] > 1:
3116 seen[label] = seen.get(label, 0) + 1
3117 labels[i] = f"{label}#{seen[label]}"
3118 return labels
3121def source_file_name(label: str) -> str:
3122 """The file stem inside a :func:`source_labels` label — without the folders
3123 that qualify it, its ``#n`` occurrence number or a kept extension — for
3124 code that parses identity out of a file's name (MultiplEYE uploads)."""
3125 name = re.sub(r"#\d+$", "", str(label).rsplit("/", 1)[-1])
3126 stem = Path(name).stem
3127 return stem if Path(name).suffix.lower().lstrip(".") in UPLOAD_FILE_TYPES else name
3130def _tag_and_concat(
3131 frames: list[pd.DataFrame],
3132 labels: list[str],
3133 source_column: str | None,
3134 *,
3135 always_tag: bool = False,
3136) -> pd.DataFrame:
3137 """Concatenate frames into one, tagging each with its source label in
3138 ``source_column`` (unless that frame already carries the column, or
3139 ``source_column`` is None) so rows stay traceable to their origin.
3141 By default only multi-frame reads are tagged (a lone frame needs no origin
3142 marker). ``always_tag=True`` tags a single frame too — used by
3143 :func:`read_tables` so a one-file upload still exposes its filename (as
3144 ``source_file``), which the upload wizard can map as the trial / participant
3145 id when no column carries it. Columns are aligned by name; fields absent
3146 from a frame become NaN for its rows."""
3147 if source_column and (always_tag or len(frames) > 1):
3148 for df, label in zip(frames, labels):
3149 if source_column not in df.columns:
3150 df[source_column] = label
3151 if len(frames) == 1:
3152 return frames[0]
3153 return pd.concat(frames, ignore_index=True, sort=False)
3156# DATA-16 (security review S6): bound zip decompression. `_read_zipped_table`
3157# used to `read()` every member with no ceiling, so a small archive of highly
3158# compressible CSV could expand to many gigabytes and OOM-kill the process — on
3159# the ~1 GB hosted demo that takes every concurrent visitor's session with it.
3160# The limits are deliberately generous: real eye-tracking exports are large (the
3161# OneStop reports are hundreds of MB to a few GB of CSV per zipped table), so the
3162# absolute caps only catch the honestly-enormous case, and the *ratio* is what
3163# actually distinguishes a zip bomb from a big corpus.
3164#
3165# DATA-34 raised the absolute caps (4/8 GB → 32/64 GB) and made all three
3166# tunable: a full OneStop fixation report is a single ~8 GB CSV inside its zip,
3167# which the old per-member cap refused outright on a workstation with the memory
3168# to read it. The caps are a memory guard, not a security boundary — the ratio
3169# check is what discriminates a bomb — so the default now suits the machine that
3170# actually holds a corpus, and a memory-capped deployment tightens it with the
3171# env vars below (read once, at import).
3172ZIP_MAX_MEMBER_ENV = "SCANPATH_ZIP_MAX_MEMBER_GB"
3173ZIP_MAX_TOTAL_ENV = "SCANPATH_ZIP_MAX_TOTAL_GB"
3174ZIP_MAX_RATIO_ENV = "SCANPATH_ZIP_MAX_RATIO"
3177def _zip_limit_from_env(var: str, default: float) -> float:
3178 """Read a zip limit from the environment, falling back to ``default``.
3180 The value is returned in whatever unit the caller works in (the two size
3181 vars are read as gigabytes and scaled here; the ratio is unitless). An
3182 unset, unparseable or non-positive value keeps the default rather than
3183 failing the import — a typo in a deployment's env should not stop the app
3184 from reading data."""
3185 raw = os.environ.get(var, "").strip()
3186 if not raw:
3187 return default
3188 try:
3189 value = float(raw)
3190 except ValueError:
3191 return default
3192 return value if value > 0 else default
3195ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES = int(
3196 _zip_limit_from_env(ZIP_MAX_MEMBER_ENV, 32.0) * 1024**3
3197) # any single member
3198ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES = int(
3199 _zip_limit_from_env(ZIP_MAX_TOTAL_ENV, 64.0) * 1024**3
3200) # across the whole archive
3201ZIP_MAX_COMPRESSION_RATIO = _zip_limit_from_env(
3202 ZIP_MAX_RATIO_ENV, 200.0
3203) # uncompressed / compressed
3204# Below this the ratio is not checked at all: a small but very repetitive table
3205# can legitimately compress 1000×, and expanding it costs nothing.
3206ZIP_RATIO_CHECK_MIN_BYTES = 256 * 1024 * 1024
3209def _format_bytes(n: float) -> str:
3210 """Human-readable byte size for a user-facing error message."""
3211 for unit in ("B", "KB", "MB", "GB"):
3212 if abs(n) < 1024 or unit == "GB":
3213 return f"{n:.0f} {unit}" if unit == "B" else f"{n:.1f} {unit}"
3214 n /= 1024.0
3215 return f"{n:.1f} GB"
3218def _check_zip_limits(infos: list[zipfile.ZipInfo]) -> None:
3219 """Reject an archive that would decompress past the DATA-16 limits.
3221 Checks the *declared* sizes (``ZipInfo.file_size``) before a single member is
3222 opened, so an oversized archive fails fast instead of being discovered by
3223 exhausting RAM. Declared sizes can be forged, so :func:`_read_zipped_table`
3224 additionally reads each member under a hard byte budget.
3225 """
3226 total = sum(int(i.file_size) for i in infos)
3227 compressed = sum(int(i.compress_size) for i in infos)
3228 for info in infos:
3229 if int(info.file_size) > ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES:
3230 raise ValueError(
3231 f"{info.filename!r} in the zip unpacks to "
3232 f"{_format_bytes(info.file_size)}, over the "
3233 f"{_format_bytes(ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES)} per-file "
3234 "limit. Split it into smaller files (one per participant works "
3235 f"well). Running it yourself? {ZIP_MAX_MEMBER_ENV} (GB) raises it."
3236 )
3237 if total > ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES:
3238 raise ValueError(
3239 f"the zip unpacks to {_format_bytes(total)}, over the "
3240 f"{_format_bytes(ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES)} limit. Split it "
3241 "into smaller uploads. Running it yourself? "
3242 f"{ZIP_MAX_TOTAL_ENV} (GB) raises it."
3243 )
3244 if total >= ZIP_RATIO_CHECK_MIN_BYTES:
3245 ratio = total / max(compressed, 1)
3246 if ratio > ZIP_MAX_COMPRESSION_RATIO:
3247 raise ValueError(
3248 f"the zip archive expands {ratio:.0f}× (to "
3249 f"{_format_bytes(total)}), above the "
3250 f"{ZIP_MAX_COMPRESSION_RATIO:.0f}× limit — it doesn't look like "
3251 "a normal data export. Unzip it and load the tables directly if "
3252 "this is genuine."
3253 )
3256#: Formats whose pandas reader seeks (columnar footers, zip-container
3257#: workbooks), so their member can't be streamed and is read into memory whole.
3258_SEEKABLE_ONLY_SUFFIXES = (".parquet", ".feather", ".xlsx", ".xls")
3261class _BudgetedZipMember(io.RawIOBase):
3262 """One zip member, readable only up to ``budget`` bytes.
3264 A zip's central directory is attacker-controlled, so the declared sizes
3265 :func:`_check_zip_limits` reads can be forged; this counts the bytes that
3266 actually arrive and raises as soon as they pass the budget. Streaming them
3267 (rather than ``read()``-ing the member into one big ``bytes`` first) also
3268 keeps an honestly enormous CSV — a full OneStop fixation report is ~8 GB
3269 inside its zip — from needing a second full-size copy of itself in memory
3270 before pandas sees a byte of it (DATA-34).
3271 """
3273 def __init__(
3274 self,
3275 inner,
3276 budget: int,
3277 name: str,
3278 *,
3279 limit_label: str = "archive decompression limit",
3280 ) -> None:
3281 self._inner = inner
3282 self._budget = int(budget)
3283 self._name = name
3284 self._limit_label = limit_label
3285 self.consumed = 0
3287 def readable(self) -> bool:
3288 return True
3290 def readinto(self, buffer) -> int:
3291 read = self._inner.readinto(buffer)
3292 if not read:
3293 return 0
3294 self.consumed += read
3295 if self.consumed > self._budget:
3296 raise ValueError(
3297 f"{self._name!r} in the zip is larger than the archive says, "
3298 f"past the {_format_bytes(self._budget)} {self._limit_label}, so "
3299 "it was not read. Re-create the zip and upload it again."
3300 )
3301 return read
3304#: What each table kind's members are called in the mixed-zip message (#374 F3).
3305_ZIP_KIND_NOUNS = {
3306 "fixations": ("fixation report", "fixation reports"),
3307 "words": ("interest-area report", "interest-area reports"),
3308}
3309#: The wizard row each kind belongs in, as the message names it.
3310_ZIP_KIND_ROWS = {"fixations": "Fixations", "words": "Words"}
3313def table_kind_of_columns(columns: Iterable[str]) -> str | None:
3314 """Whether a header looks like a fixation table or a word (interest-area)
3315 table — ``"fixations"``, ``"words"`` or ``None`` when it is neither or both.
3317 Reuses the mapping's own detection: a fixation table has a duration column
3318 and no word box, a word table has a word box and no fixation duration. An
3319 EyeLink Fixation Report and Interest Area Report land on opposite sides.
3320 """
3321 frame = pd.DataFrame(columns=list(dict.fromkeys(map(str, columns))))
3322 words = propose_word_schema(frame)
3323 has_box = all(words.get(k) for k in ("x", "y", "width", "height")) or all(
3324 words.get(k) for k in _BOX_EDGES
3325 )
3326 has_duration = bool(propose_fix_schema(frame).get("duration"))
3327 if has_duration and not has_box:
3328 return "fixations"
3329 if has_box and not has_duration:
3330 return "words"
3331 return None
3334@dataclass(frozen=True)
3335class ZipMemberSplit:
3336 """Which members of a zip one table reads (#374 F3).
3338 A ZIP of per-participant folders commonly holds both EyeLink reports. Its
3339 members are grouped by column set, each set classified by
3340 :func:`table_kind_of_columns`; when the archive holds both kinds, the table
3341 being read keeps the members of its own kind and the rest are *left out* —
3342 never concatenated into one frame, which put a "fixation" at every word.
3343 """
3345 used: tuple[str, ...]
3346 left_out: tuple[str, ...] = ()
3347 kind: str | None = None
3348 left_out_kind: str | None = None
3349 used_infos: tuple = ()
3351 @property
3352 def mixed(self) -> bool:
3353 return bool(self.left_out)
3355 def message(self) -> str:
3356 """The one-line note the upload row shows ("" when nothing was left out)."""
3357 if not self.left_out or self.kind not in _ZIP_KIND_NOUNS:
3358 return ""
3359 n_used, n_out = len(self.used), len(self.left_out)
3360 noun = _ZIP_KIND_NOUNS[self.kind][n_used != 1]
3361 if self.left_out_kind in _ZIP_KIND_NOUNS:
3362 other = _ZIP_KIND_NOUNS[self.left_out_kind][n_out != 1]
3363 row = _ZIP_KIND_ROWS[self.left_out_kind]
3364 tail = f" — add the ZIP to the {row} row too."
3365 else:
3366 other, tail = ("other file" if n_out == 1 else "other files"), "."
3367 verb = "was" if n_out == 1 else "were"
3368 return (
3369 f"Using the {n_used} {noun} in this ZIP; {n_out} {other} {verb} "
3370 f"left out{tail}"
3371 )
3374def _zip_member_columns(zf: zipfile.ZipFile, info: zipfile.ZipInfo) -> list[str]:
3375 """One member's column names — the header line for delimited text, else
3376 the member read (under the per-file budget) and its schema taken."""
3377 name = info.filename.lower()
3378 if name.endswith((".tsv", ".tab", ".csv", ".txt")):
3379 return _member_layout(zf, info)[0]
3380 with zf.open(info) as inner:
3381 stream = _BudgetedZipMember(
3382 inner, ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES, info.filename
3383 )
3384 buf = io.BytesIO(stream.read())
3385 buf.name = info.filename
3386 return read_table_columns(buf)
3389def _zip_split(
3390 zf: zipfile.ZipFile, infos: list[zipfile.ZipInfo], kind: str | None
3391) -> ZipMemberSplit:
3392 """Split ``infos`` into the members ``kind`` reads and those it leaves out."""
3393 every = ZipMemberSplit(
3394 used=tuple(i.filename for i in infos), kind=kind, used_infos=tuple(infos)
3395 )
3396 if kind not in _ZIP_KIND_NOUNS or len(infos) < 2:
3397 return every
3398 by_columns: dict[tuple, str | None] = {}
3399 kinds = []
3400 for info in infos:
3401 columns = tuple(_zip_member_columns(zf, info))
3402 if columns not in by_columns:
3403 by_columns[columns] = table_kind_of_columns(columns)
3404 kinds.append(by_columns[columns])
3405 present = {k for k in kinds if k}
3406 if len(present) < 2 or kind not in present:
3407 return every
3408 used = [i for i, k in zip(infos, kinds) if k == kind]
3409 out_kinds = {k for k in kinds if k != kind}
3410 return ZipMemberSplit(
3411 used=tuple(i.filename for i in used),
3412 left_out=tuple(i.filename for i, k in zip(infos, kinds) if k != kind),
3413 kind=kind,
3414 left_out_kind=out_kinds.pop() if len(out_kinds) == 1 else None,
3415 used_infos=tuple(used),
3416 )
3419def zip_member_split(file_like_or_path, kind: str | None) -> ZipMemberSplit:
3420 """Which members of a ``.zip`` the ``kind`` table reads, and which it
3421 leaves out (#374 F3) — what the upload row reports. A non-zip, or a zip
3422 holding one kind of table, uses everything."""
3423 name = getattr(file_like_or_path, "name", str(file_like_or_path))
3424 if not name.lower().endswith(".zip"):
3425 return ZipMemberSplit(used=(name,), kind=kind)
3426 _rewind(file_like_or_path)
3427 try:
3428 with zipfile.ZipFile(file_like_or_path) as zf:
3429 infos = [
3430 i
3431 for i in zf.infolist()
3432 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__"))
3433 ]
3434 _check_zip_limits(infos)
3435 return _zip_split(zf, infos, kind)
3436 finally:
3437 _rewind(file_like_or_path)
3440#: Rows :func:`read_table_sample` parses, and the bytes it reads of a zip member.
3441_SAMPLE_ROWS = 2000
3442_SAMPLE_BYTES = 4 * 1024 * 1024
3445def read_table_sample(
3446 file_like_or_path, *, kind: str | None = None, nrows: int = _SAMPLE_ROWS
3447) -> pd.DataFrame:
3448 """The first rows of a delimited table, every column parsed (#374 F13).
3450 What the wizard judges an unmapped column's values by, when its planned
3451 read (PERF-6) left that column out. Delimited text only — a plain file or
3452 a zip's first member of ``kind`` (:func:`zip_member_split`); anything else,
3453 or a file that will not parse, gives an empty frame, which the caller reads
3454 as "no evidence".
3455 """
3456 name = getattr(file_like_or_path, "name", str(file_like_or_path)).lower()
3457 delimited = (".tsv", ".tab", ".csv", ".txt")
3458 _rewind(file_like_or_path)
3459 try:
3460 if name.endswith(delimited):
3461 sep = _sniff_delimiter(file_like_or_path, name)
3462 return _read_delimited(file_like_or_path, sep, None, nrows=nrows)
3463 if not name.endswith(".zip"):
3464 return pd.DataFrame()
3465 with zipfile.ZipFile(file_like_or_path) as zf:
3466 infos = [
3467 i
3468 for i in zf.infolist()
3469 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__"))
3470 ]
3471 _check_zip_limits(infos)
3472 used = _zip_split(zf, infos, kind).used_infos
3473 info = next(
3474 (i for i in used if i.filename.lower().endswith(delimited)), None
3475 )
3476 if info is None:
3477 return pd.DataFrame()
3478 sep = _member_layout(zf, info)[1]
3479 with zf.open(info) as inner:
3480 head = inner.read(_SAMPLE_BYTES)
3481 # Cut at the last whole line: the read stops mid-row.
3482 if len(head) == _SAMPLE_BYTES and b"\n" in head:
3483 head = head[: head.rindex(b"\n") + 1]
3484 return _read_delimited(io.BytesIO(head), sep, None, nrows=nrows)
3485 except Exception: # no evidence, not an error: the real read reports it
3486 return pd.DataFrame()
3487 finally:
3488 _rewind(file_like_or_path)
3491def looks_like_condition(values: pd.Series) -> bool:
3492 """Whether a column reads as a condition or an item id — a few repeated
3493 values, such as ``Adv`` / ``Ele`` or ``2_1`` … ``2_12`` (#374 F13).
3495 Between 2 and 50 distinct values, each used at least twice on average; a
3496 fractional number is a measurement, never a condition."""
3497 filled = values.dropna()
3498 if filled.empty:
3499 return False
3500 if pd.api.types.is_float_dtype(filled) and not (filled % 1 == 0).all():
3501 return False
3502 distinct = filled.nunique()
3503 return 2 <= distinct <= 50 and distinct * 2 <= len(filled)
3506def _read_zipped_table(
3507 file_like_or_path, *, plan: ReadPlan | None = None, kind: str | None = None
3508) -> pd.DataFrame:
3509 """Read table(s) from a ``.zip`` archive (e.g. ``data.csv.zip``).
3511 Each member is dispatched on its own extension, so a zip may wrap any
3512 supported format. A multi-member archive is concatenated just like a
3513 multi-file upload — every member's rows tagged with its stem in
3514 ``source_file`` (qualified by its folders when two members share a stem,
3515 :func:`source_labels`). With ``kind``, an archive that mixes fixation and
3516 interest-area reports keeps only the members that fit it
3517 (:func:`zip_member_split`, #374 F3). pandas infers compression only from string paths, not from
3518 uploaded file-like objects, so we open the archive ourselves. Raises
3519 ``ValueError`` if the archive holds no data file (macOS ``__MACOSX``/dotfile
3520 cruft is ignored), or if it would decompress past the DATA-16 size limits
3521 (``ZIP_MAX_*``) — both the declared sizes and the bytes actually read are
3522 bounded, so a forged header can't slip past."""
3523 with zipfile.ZipFile(file_like_or_path) as zf:
3524 infos = [
3525 i
3526 for i in zf.infolist()
3527 if not i.is_dir() and not Path(i.filename).name.startswith((".", "__"))
3528 ]
3529 if not infos:
3530 raise ValueError(
3531 "the zip holds no table file (CSV, TSV, TXT, Excel, Parquet or Feather)"
3532 )
3533 _check_zip_limits(infos)
3534 infos = _zip_split(zf, infos, kind).used_infos
3535 remaining = ZIP_MAX_TOTAL_UNCOMPRESSED_BYTES
3536 frames, labels = [], []
3537 for info in infos:
3538 member = info.filename
3539 name = member.lower()
3540 member_budget = min(remaining, ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES)
3541 limit_label = (
3542 "per-file limit"
3543 if ZIP_MAX_MEMBER_UNCOMPRESSED_BYTES <= remaining
3544 else "archive decompression limit"
3545 )
3546 with zf.open(info) as inner:
3547 stream = _BudgetedZipMember(
3548 inner,
3549 member_budget,
3550 member,
3551 limit_label=limit_label,
3552 )
3553 if name.endswith(_SEEKABLE_ONLY_SUFFIXES):
3554 # Columnar/workbook readers seek, so these still land in
3555 # memory whole — bounded by the same budget.
3556 buf = io.BytesIO(stream.read())
3557 # BUG-84: the header pass dispatches on the name, and an
3558 # unnamed buffer read its binary member as CSV.
3559 buf.name = member
3560 member_plan = plan
3561 if plan is not None and plan.columns:
3562 member_plan = plan.narrowed_to(read_table_columns(buf))
3563 buf.seek(0)
3564 frames.append(_read_by_extension(buf, name, member_plan))
3565 else:
3566 # PERF-6: one archive can hold members with different
3567 # columns, and the plan is built from their union — narrow
3568 # it to this member's own header or `usecols` rejects it.
3569 header, sep = _member_layout(zf, info)
3570 member_plan = plan
3571 if plan is not None and plan.columns:
3572 member_plan = plan.narrowed_to(header)
3573 try:
3574 frames.append(
3575 _read_by_extension(
3576 io.BufferedReader(stream), name, member_plan, sep=sep
3577 )
3578 )
3579 except UnicodeDecodeError:
3580 # BUG-55: not UTF-8, and a member stream cannot be
3581 # rewound for the encoding fallback — read it again
3582 # into memory, under the same budget, where it can.
3583 with zf.open(info) as again:
3584 stream = _BudgetedZipMember(
3585 again, member_budget, member, limit_label=limit_label
3586 )
3587 buf = io.BytesIO(stream.read())
3588 frames.append(
3589 _read_by_extension(buf, name, member_plan, sep=sep)
3590 )
3591 remaining -= stream.consumed
3592 labels.append(member)
3593 return _tag_and_concat(frames, source_labels(labels), SOURCE_FILE_COLUMN)
3596# BUG-5: guard the memory-constrained hosted demo against a too-large upload.
3597# The bytes upload fine; the OOM comes later, when a big (often zipped) table
3598# decompresses and pandas holds several copies through parse + normalization —
3599# on Streamlit Community Cloud (~1 GB RAM) that silently kills the process with
3600# no traceback. Above this raw-upload size the wizard warns and asks for an
3601# explicit opt-in before parsing (a one-click confirm locally, real protection on
3602# the host). Tuned to sit below the OneStop repeated-reading export
3603# (~29–37 MB per zipped table) that first surfaced this.
3604UPLOAD_SIZE_WARN_BYTES = 25 * 1024 * 1024
3607def uploaded_files_total_bytes(uploaded) -> int:
3608 """Total byte size of a Streamlit upload — one ``UploadedFile`` or a list.
3610 Reads the ``.size`` each file already carries (no data copy). ``None`` /
3611 empty → 0, so an absent upload is trivially under any threshold."""
3612 if not uploaded:
3613 return 0
3614 files = uploaded if isinstance(uploaded, (list, tuple)) else [uploaded]
3615 return sum(int(getattr(f, "size", 0) or 0) for f in files)
3618def upload_exceeds_limit(
3619 uploaded, threshold_bytes: int = UPLOAD_SIZE_WARN_BYTES
3620) -> bool:
3621 """Whether a Streamlit upload's total size is over the guard threshold (BUG-5)."""
3622 return uploaded_files_total_bytes(uploaded) > threshold_bytes
3625def read_table(
3626 file_like_or_path, *, plan: ReadPlan | None = None, kind: str | None = None
3627) -> pd.DataFrame:
3628 """Read a tabular file by extension: csv, tsv, parquet, feather, or a
3629 ``.zip`` wrapping one or more of those (e.g. ``data.csv.zip``). A
3630 multi-member zip is concatenated like a multi-file upload.
3632 ``plan`` (PERF-6) is a :class:`ReadPlan` from :func:`plan_table_read`,
3633 narrowing the read to the columns normalization keeps. ``kind``
3634 (``"words"`` / ``"fixations"``) is the table being read: a zip that mixes
3635 the two keeps only the members that fit it (:func:`zip_member_split`)."""
3636 name = getattr(file_like_or_path, "name", str(file_like_or_path)).lower()
3637 try:
3638 if name.endswith(".zip"):
3639 return _read_zipped_table(file_like_or_path, plan=plan, kind=kind)
3640 return _read_by_extension(file_like_or_path, name, plan)
3641 except pd.errors.EmptyDataError as exc:
3642 raise _empty_file_error(name) from exc
3645def expand_table_inputs(inputs: TablesInput) -> list:
3646 """Flatten a path / glob pattern / file-like / list-of-those into a list.
3648 Glob patterns are expanded in sorted order so multi-file datasets (one
3649 file per participant or per stimulus) can be referenced with a single
3650 pattern like ``scanpaths/*.tsv``. Raises ``FileNotFoundError`` for a
3651 pattern that matches nothing — silently loading zero files would read as
3652 success."""
3653 if not isinstance(inputs, (list, tuple)):
3654 inputs = [inputs]
3655 expanded: list = []
3656 for item in inputs:
3657 if isinstance(item, (str, os.PathLike)) and glob.has_magic(str(item)):
3658 matches = sorted(glob.glob(str(item), recursive=True))
3659 if not matches:
3660 raise FileNotFoundError(f"No files match pattern: {item}")
3661 expanded.extend(matches)
3662 else:
3663 expanded.append(item)
3664 return expanded
3667def read_tables(
3668 inputs: TablesInput,
3669 source_column: str | None = SOURCE_FILE_COLUMN,
3670 *,
3671 plan_for=None,
3672 kind: str | None = None,
3673) -> pd.DataFrame:
3674 """Read one or many tabular files and concatenate them into one frame.
3676 ``inputs`` may be a single path or file-like object, a glob pattern, or a
3677 list mixing those (a ``.zip`` member counts as a file too). ``plan_for`` is
3678 called with each file's column names and returns the :class:`ReadPlan` to
3679 read it under (PERF-6); omit it to parse every column. Each part gets a
3680 ``source_file`` column holding the file's stem — qualified by its folders
3681 when two files share one (:func:`source_labels`) — (unless the data already has
3682 that column, or ``source_column=None``) — *including a single file*, so
3683 datasets that key identity in the filename can recover it (the upload wizard
3684 maps ``source_file`` as the trial / participant id). Columns are aligned by
3685 name across files; fields absent from a file become NaN for its rows.
3686 ``kind`` picks a mixed zip's members (:func:`zip_member_split`)."""
3687 items = expand_table_inputs(inputs)
3688 frames, labels = [], []
3689 for item in items:
3690 # PERF-6: each file is planned against its OWN header. One file per
3691 # participant is the common upload shape, and an export can gain or
3692 # lose a column between them — a shared plan would name a column some
3693 # file hasn't got, which `usecols` raises on.
3694 plan = (
3695 plan_for(read_table_columns(item, kind=kind))
3696 if plan_for is not None
3697 else None
3698 )
3699 frames.append(read_table(item, plan=plan, kind=kind))
3700 labels.append(getattr(item, "name", str(item)))
3701 return _tag_and_concat(
3702 frames, source_labels(labels), source_column, always_tag=True
3703 )
3706def _load_bundled(name: str) -> pd.DataFrame:
3707 """Load a single bundled sample, preferring Parquet over CSV."""
3708 data_root = resources.files(PACKAGE_NAME).joinpath("sample_data")
3709 for ext in (".parquet", ".csv"):
3710 resource = data_root / f"{name}{ext}"
3711 try:
3712 with resources.as_file(resource) as path:
3713 if not path.is_file():
3714 continue
3715 return read_table(path)
3716 except FileNotFoundError:
3717 continue
3718 return pd.DataFrame()
3721def _resolve_sample_image_paths(df: pd.DataFrame) -> pd.DataFrame:
3722 """Expand the bundled demo's relative ``image_path`` (e.g.
3723 ``images/2_2_1_Adv__paragraph.png``) into an absolute path under the
3724 packaged ``sample_data`` directory, so the stimulus-image background layer
3725 (``tabs._render_single_trial`` → ``plots.make_scanpath_figure``) can load it
3726 via ``os.path.exists``. The CSVs ship a *relative* reference (stable across
3727 installs); this resolves it at load time against wherever the wheel landed.
3728 No-op when the column is absent or a value is already absolute."""
3729 if "image_path" not in df.columns:
3730 return df
3731 try:
3732 root = Path(str(resources.files(PACKAGE_NAME).joinpath("sample_data")))
3733 except (ModuleNotFoundError, FileNotFoundError, TypeError):
3734 return df
3736 def _abs(value: object) -> object:
3737 if isinstance(value, str) and value and not os.path.isabs(value):
3738 return str(root.joinpath(*value.split("/")))
3739 return value
3741 df = df.copy()
3742 df["image_path"] = df["image_path"].map(_abs)
3743 return df
3746def _pattern_placeholders(pattern: str) -> list[str]:
3747 """The row fields a filename pattern reads, in first-appearance order.
3749 ``"{text_id}/{trial_id:0>3}.png"`` reads ``text_id`` and ``trial_id``.
3750 An attribute or index suffix is trimmed back to the field the row is keyed
3751 by (``{a.b}`` reads ``a``), matching the row-dict lookup this feeds.
3752 Auto-numbered fields (``{}``) name no row field and are skipped —
3753 ``format_map`` raises for them whether or not they are collected.
3754 """
3755 names: list[str] = []
3756 for _, field_name, _, _ in string.Formatter().parse(pattern):
3757 if not field_name:
3758 continue
3759 name = re.split(r"[.\[]", field_name, maxsplit=1)[0]
3760 if name and name not in names:
3761 names.append(name)
3762 return names
3765def resolve_stimulus_image_paths(
3766 frame: pd.DataFrame,
3767 root: str | os.PathLike,
3768 pattern: str = "{text_id}.png",
3769 *,
3770 require_exists: bool = True,
3771) -> pd.DataFrame:
3772 """Attach per-row stimulus images from a local folder and filename pattern.
3774 Placeholders are read from the row (for example ``{text_id}``,
3775 ``{trial_id}``, or ``{participant_id}``). Relative subdirectories are
3776 supported, but resolved files must remain under ``root``; absolute and
3777 parent-traversal patterns are rejected. Rows whose placeholders are missing
3778 or whose file does not exist keep their previous ``image_path`` value.
3780 This pure helper is the headless/API surface for VIZ-14 and is also used by
3781 the desktop-only folder controls and the CLI. It is deliberately
3782 **undecorated**: the CLI and the API have no Streamlit session to cache
3783 into, so any caching belongs at a call site rather than here.
3785 **One filesystem probe per distinct placeholder tuple, not per row.** A
3786 pattern reads a couple of columns, so a fixation frame of millions of rows
3787 still resolves only a few hundred distinct paths — where the per-row form
3788 this replaced spent two syscalls each (``Path.resolve`` + ``is_file``) on
3789 *every* rerun that had a folder attached. Per-row behaviour is unchanged:
3790 the fallback is each row's **own** prior ``image_path``, not a shared one,
3791 and a NaN placeholder is still absent from the format mapping rather than
3792 the string ``"nan"`` (which would otherwise resolve ``nan.png`` — a miss
3793 today, a wrong hit the day such a file exists).
3794 """
3795 if frame is None or frame.empty:
3796 return frame.copy() if isinstance(frame, pd.DataFrame) else pd.DataFrame()
3797 base = Path(root).expanduser().resolve()
3798 if not pattern or Path(pattern).is_absolute():
3799 raise ValueError("Image filename pattern must be a non-empty relative path.")
3801 class _Row(dict):
3802 def __missing__(self, key):
3803 raise KeyError(key)
3805 def _resolved(values: dict[str, str]) -> str | None:
3806 """One placeholder tuple to an absolute path, or None to keep `previous`."""
3807 try:
3808 relative = pattern.format_map(_Row(values))
3809 except (KeyError, ValueError, AttributeError):
3810 return None
3811 candidate = (base / relative).resolve()
3812 try:
3813 candidate.relative_to(base)
3814 except ValueError as exc:
3815 raise ValueError(
3816 f"Image pattern resolves outside the selected folder: {relative!r}"
3817 ) from exc
3818 if require_exists and not candidate.is_file():
3819 return None
3820 return str(candidate)
3822 # Keyed by the *stringified* column label, as the per-row dict was; a
3823 # duplicate label keeps the last column, as that dict's later write did.
3824 by_name: dict[str, object] = {str(column): column for column in frame.columns}
3825 used = [by_name[name] for name in _pattern_placeholders(pattern) if name in by_name]
3827 if used:
3828 text = pd.DataFrame(index=frame.index)
3829 missing = np.zeros(len(frame), dtype=bool)
3830 for position, column in enumerate(used):
3831 values = frame[column]
3832 if isinstance(values, pd.DataFrame): # duplicate column labels
3833 values = values.iloc[:, -1]
3834 missing |= values.isna().to_numpy()
3835 text[position] = values.astype(str)
3836 codes, uniques = pd.factorize(pd.MultiIndex.from_frame(text))
3837 names = [str(column) for column in used]
3838 resolved = np.asarray(
3839 [_resolved(dict(zip(names, key))) for key in uniques], dtype=object
3840 )[codes]
3841 # A NaN placeholder never reached the format mapping, so the row keeps
3842 # its own `image_path`; `astype(str)` above turned it into "nan".
3843 resolved[missing] = None
3844 else:
3845 resolved = np.full(len(frame), _resolved({}), dtype=object)
3847 series = pd.Series(resolved, index=frame.index, dtype=object)
3848 if "image_path" in frame.columns:
3849 series = series.where(series.notna(), frame["image_path"])
3850 result = frame.copy()
3851 result["image_path"] = series
3852 return result
3855@st.cache_data
3856def load_sample_data() -> tuple[pd.DataFrame, pd.DataFrame]:
3857 """Load bundled demo IA and fixation tables (prefer Parquet).
3859 The tables ship per-trial stimulus-image references (``image_path`` +
3860 ``image_x``/``image_y`` origin) pointing at the rendered paragraph PNGs
3861 under ``sample_data/images/``; the relative paths are resolved to absolute
3862 here so the optional stimulus-image background layer renders the page
3863 behind the scanpath (aligned to the 2560x1440 OneStop monitor)."""
3864 words = _resolve_sample_image_paths(_load_bundled("ia"))
3865 fixations = _resolve_sample_image_paths(_load_bundled("fixations"))
3866 if words.empty or fixations.empty:
3867 st.error(
3868 "The bundled demo is missing from this installation. Reinstall "
3869 "Scanpath Studio, then reload the page."
3870 )
3871 return pd.DataFrame(), pd.DataFrame()
3872 return words, fixations
3875@st.cache_data
3876def load_sample_raw_gaze() -> pd.DataFrame:
3877 """Load bundled raw gaze sample (millisecond-level x,y)."""
3878 return _load_bundled("raw_gaze")
3881def infer_raw_gaze_schema(raw_gaze: pd.DataFrame) -> dict[str, str] | None:
3882 """Infer schema for raw millisecond-level gaze data."""
3883 schema = propose_raw_gaze_schema(raw_gaze)
3884 problems = validate_raw_gaze_schema(schema)
3885 if problems:
3886 st.error(f"Missing required raw gaze fields: {', '.join(problems)}")
3887 return None
3888 return schema
3891def normalize_raw_gaze(
3892 raw_gaze: pd.DataFrame, schema: dict[str, str], *, keep_columns: set | None = None
3893) -> pd.DataFrame:
3894 """Normalize raw gaze data to canonical column names.
3896 ``keep_columns`` (UX-120) carries user-chosen extra source columns
3897 through verbatim, the same "Extra fields to keep" mechanism
3898 :func:`normalize_words`/:func:`normalize_fixations` already have — raw
3899 gaze has no optional-fields *registry* of its own (no semantic field
3900 here is common enough across exports to earn a canonical name the way
3901 ``saccade_amplitude`` does for fixations), so this only ever carries
3902 unclaimed columns, via :func:`_carry_extra_columns`.
3903 """
3904 # Samples with no participant or trial id belong to no trial (BUG-56's
3905 # rule for the other two tables, round 10).
3906 blank, unkeyed = _rows_missing_identity(raw_gaze, schema)
3907 for issue in identity_issues(raw_gaze, schema, table="Raw gaze", unkeyed=unkeyed):
3908 warnings.warn(issue.replace("`", "'"), UserWarning, stacklevel=2)
3909 if (blank | unkeyed).any():
3910 raw_gaze = raw_gaze.loc[~(blank | unkeyed)]
3911 df = pd.DataFrame(index=raw_gaze.index)
3912 if schema.get("participant"):
3913 # str or list (a composite participant id), joined like normalize_words —
3914 # so a composite participant stays key-compatible with the fixations.
3915 df["participant_id"] = trial_id_series(raw_gaze, schema["participant"])
3916 else:
3917 df["participant_id"] = SYNTHETIC_PARTICIPANT
3918 trial_cols = trial_mapping_columns(schema["trial"])
3919 if len(trial_cols) > 1:
3920 # User-composed unique trial ID — see normalize_words.
3921 df["trial_id"] = trial_id_series(raw_gaze, trial_cols)
3922 df["unique_trial_id"] = df["trial_id"]
3923 else:
3924 trial_col = trial_cols[0] # the mapped column (BUG-58)
3925 df["trial_id"] = stable_id(raw_gaze[trial_col])
3926 if "unique_trial_id" in raw_gaze.columns:
3927 # The mapped id *is* the unique trial id (BUG-58) — never the raw
3928 # column's own values, which the trial picker would otherwise key
3929 # on (`utils.build_combo_options` prefers `unique_trial_id`).
3930 df["unique_trial_id"] = df["trial_id"]
3931 # UX-113: mapped when the export carries its own text/passage column;
3932 # otherwise raw gaze has no text/passage concept of its own, so mirror
3933 # trial_id — a raw-gaze-only dataset still needs *a* text_id column for
3934 # the trial picker (utils.build_combo_options).
3935 if schema.get("text_id"):
3936 df["text_id"] = raw_gaze[schema["text_id"]].astype(str)
3937 else:
3938 df["text_id"] = df["trial_id"]
3939 df = _copy_screen_fields(df, raw_gaze, schema)
3940 if schema.get("text"):
3941 df["text"] = raw_gaze[schema["text"]].astype(str)
3942 else:
3943 df["text"] = ""
3944 # UX-113: raw gaze samples are never run through word assignment the way
3945 # fixations are (there is no per-sample geometry step for it) — carried
3946 # through only when the export already names one, not computed.
3947 if schema.get("word_id"):
3948 df["word_id"] = raw_gaze[schema["word_id"]]
3949 df["x"] = _to_number(raw_gaze[schema["x"]])
3950 df["y"] = _to_number(raw_gaze[schema["y"]])
3951 skip = _schema_source_columns(schema)
3952 if schema.get("timestamp"):
3953 onset = schema["timestamp"]
3954 df["timestamp_ms"] = _as_ms(_to_number(raw_gaze[onset]), onset) # DATA-40
3955 else:
3956 # No clock: the samples keep their order (1, 2, … per trial) and no
3957 # time. Nothing says how far apart they were recorded, so a made-up
3958 # `timestamp_ms` would be a sampling rate the data never stated — and
3959 # it would reach the plot's colour scale, hover and the exports as ms.
3960 df[SAMPLE_INDEX] = df.groupby(list(PARENT_KEY), sort=False).cumcount() + 1
3961 # A kept extra named `timestamp_ms` would be read as that clock again.
3962 skip = skip | {"timestamp_ms"}
3963 if keep_columns is not None:
3964 _carry_extra_columns(df, raw_gaze, keep_columns, skip)
3965 df = _preserve_composite_columns(df, raw_gaze, schema["trial"])
3966 return df
3969def infer_word_schema(words: pd.DataFrame) -> dict[str, str] | None:
3970 schema = propose_word_schema(words)
3971 problems = validate_word_schema(schema)
3972 if problems:
3973 st.error(f"Words table problems: {'; '.join(problems)}")
3974 return None
3975 return schema
3978def infer_fix_schema(fixations: pd.DataFrame) -> dict[str, str] | None:
3979 schema = propose_fix_schema(fixations)
3980 problems = validate_fix_schema(schema)
3981 if problems:
3982 st.error(f"Fixations schema problems: {'; '.join(problems)}")
3983 return None
3984 return schema
3987# Placeholder participant id for stimulus-level word/AoI tables (no participant
3988# column — one row per word per text, shared by every reading). The marker
3989# column flags the frame so broadcast_stimulus_words() knows to expand it.
3990STIMULUS_PARTICIPANT = ""
3991STIMULUS_WORDS_FLAG = "_stimulus_words"
3993# Synthetic participant id used when a dataset has no participant column at all
3994# (a single anonymous reader). Distinct from STIMULUS_PARTICIPANT ("") so it
3995# never collides with the stimulus-word broadcast machinery — participant_id is
3996# always present downstream (combos/filters/annotations/export/measures groupby),
3997# and the UI hides the participant selector when there's only this one value.
3998SYNTHETIC_PARTICIPANT = "(all)"
4000#: The two keys a stimulus-level AOI table can attach to a reading through
4001#: (DATA-49), as the wizard names them.
4002STIMULUS_JOIN_LABELS = {"trial_id": "Trial ID", "text_id": "Text ID"}
4003#: The mapped trial id a repeated reading had before
4004#: `_disambiguate_repeated_readings` suffixed it with `_r2`, `_r3` … — on the
4005#: fixations only, and only when a suffix was given. The stimulus join reads it
4006#: so a repeat finds the boxes of what it re-read (BUG-57, DATA-49).
4007BASE_TRIAL_ID = "_base_trial_id"
4008#: On a broadcast words copy: the AOI table's own trial id it was copied from
4009#: (the reading supplies `trial_id`). ✏️ Edit dataset keeps one copy per value
4010#: of it to get the stimulus table back (`remap_normalized_frame`).
4011AOI_TRIAL_ID = "_aoi_trial_id"
4012#: Whether this frame's `text_id` came from a mapped Text ID (a schema pick,
4013#: auto-detected or not, or a literal `unique_paragraph_id`) rather than the
4014#: trial-id fallback — written by `normalize_*` from the schema, so a mapped
4015#: Text ID whose values happen to equal the trial ids still counts (DATA-49).
4016TEXT_ID_MAPPED = "_text_id_mapped"
4017#: On a fixation: its `timestamp_ms` was made up by `normalize_fixations`
4018#: because the table mapped no onset — the reading order 0, 1, 2, …, kept so
4019#: fixations still sort, but not a time. Anything that needs elapsed time
4020#: (the summaries' reading time and speed, the replay clock) lays the
4021#: fixations end to end by their durations instead and says it is an estimate.
4022TIMESTAMP_SYNTHESIZED = "_timestamp_synthesized"
4023#: Bookkeeping columns the pipeline needs and the user never sees: kept in the
4024#: frames and the recovery cache, dropped from exports and the Data page's
4025#: tables (`drop_internal_columns`), and never offered as a field — their
4026#: leading underscore is what the field listers skip.
4027INTERNAL_COLUMNS = frozenset(
4028 {
4029 STIMULUS_WORDS_FLAG,
4030 BASE_TRIAL_ID,
4031 AOI_TRIAL_ID,
4032 TEXT_ID_MAPPED,
4033 TIMESTAMP_SYNTHESIZED,
4034 }
4035)
4038def timestamps_synthesized(fixations: pd.DataFrame | None) -> bool:
4039 """Whether any of these fixations has a made-up ``timestamp_ms``.
4041 True when normalization had no onset column to read and numbered the
4042 fixations instead (:data:`TIMESTAMP_SYNTHESIZED`). A frame without the
4043 column — one built by hand, or stored before it existed — counts as
4044 recorded, which is what it always did."""
4045 if fixations is None or TIMESTAMP_SYNTHESIZED not in fixations.columns:
4046 return False
4047 return bool(fixations[TIMESTAMP_SYNTHESIZED].fillna(False).astype(bool).any())
4050#: Scratch column the stimulus broadcast merges through.
4051_STIMULUS_KEY = "_stimulus_key"
4054def user_columns(frame: pd.DataFrame) -> list:
4055 """``frame``'s columns a user can pick — every one but :data:`INTERNAL_COLUMNS`.
4057 The one list every widget offering a frame's fields should draw from, so a
4058 bookkeeping column can never surface as a hover field, a Q&A field, a
4059 comparison field or a mapping option (DATA-49)."""
4060 return [c for c in frame.columns if c not in INTERNAL_COLUMNS]
4063def drop_internal_columns(frame: pd.DataFrame) -> pd.DataFrame:
4064 """``frame`` without :data:`INTERNAL_COLUMNS` (the same object if none)."""
4065 present = [c for c in INTERNAL_COLUMNS if c in frame.columns]
4066 return frame.drop(columns=present) if present else frame
4069def shareable_frame(frame: pd.DataFrame) -> pd.DataFrame:
4070 """``frame`` as it leaves the app: no bookkeeping, no made-up clock.
4072 :func:`drop_internal_columns`, and also ``timestamp_ms`` when normalization
4073 numbered the fixations itself (:func:`timestamps_synthesized`): those
4074 numbers are a sort order, and a file that carried them under that name
4075 would read back as recorded milliseconds."""
4076 if timestamps_synthesized(frame) and "timestamp_ms" in frame.columns:
4077 frame = frame.drop(columns="timestamp_ms")
4078 return drop_internal_columns(frame)
4081class StimulusJoinError(ValueError):
4082 """A stimulus-level AOI table that the readings in the fixations can't use.
4084 Raised by :func:`broadcast_stimulus_words` — so by ``harmonize_frames`` and
4085 every loader built on it — instead of returning a dataset with no word
4086 boxes (or, on a multipart dataset, with screens that have none): the
4087 add-dataset wizard and ✏️ Edit dataset block on it, the headless API raises
4088 it, and the CLI prints it (DATA-49)."""
4091class StimulusJoinWarning(UserWarning):
4092 """Some readings found no stimulus-level word boxes (DATA-49); the rest did.
4094 A ``UserWarning``, like the normalizers' load warnings, so the API and the
4095 CLI see it; its own class so ✏️ Edit dataset can catch it during a save and
4096 show it on the page."""
4099def _count(n: int, singular: str, plural: str) -> str:
4100 return f"{n:,} {singular if n == 1 else plural}"
4103@dataclass(frozen=True)
4104class StimulusJoin:
4105 """How a stimulus-level AOI table attached to the readings (DATA-49).
4107 A *reading* is one ``(participant_id, trial_id)`` of the fixations — one
4108 ``(participant_id, trial_id, screen_id)`` on a multipart dataset, where the
4109 counts are of *reading screens*. Each takes the boxes of the AOI table's
4110 trial it is matched to by the first of these that finds one: its own trial
4111 id, the trial id it had before a repeat's ``_r2`` suffix
4112 (``_base_trial_id``), or its Text ID. ``by_trial`` were matched the first
4113 two ways, ``by_text`` the third. A trial-id match always stands; when both
4114 tables map a real Text ID and the reading's disagrees with the matched AOI
4115 trial's, it is counted in ``text_mismatches`` (``mismatch_example`` =
4116 ``(reading's, AOI trial's)``) and warned about, never redirected.
4117 ``ambiguous_texts`` are Text IDs the AOI table gives to more than one of its
4118 trials; ``ambiguous_readings`` of the unmatched readings name one of them,
4119 and ``missing_screen_readings`` match an AOI trial that has no boxes for
4120 their screen (``missing_screens``, examples)."""
4122 readings: int
4123 by_trial: int
4124 by_text: int
4125 ambiguous_texts: tuple[str, ...] = ()
4126 ambiguous_readings: int = 0
4127 missing_screen_readings: int = 0
4128 missing_screens: tuple[str, ...] = ()
4129 text_mismatches: int = 0
4130 mismatch_example: tuple[str, str] | None = None
4131 multipart: bool = False
4133 @property
4134 def matched(self) -> int:
4135 """Readings (reading screens) that got word boxes."""
4136 return self.by_trial + self.by_text
4138 @property
4139 def unmatched(self) -> int:
4140 return self.readings - self.matched
4142 @property
4143 def needs_warning(self) -> bool:
4144 """Whether :meth:`describe` reports something to act on: readings with
4145 no boxes, or readings whose Text IDs disagree with their boxes'."""
4146 return bool(self.unmatched or self.text_mismatches)
4148 @property
4149 def key(self) -> str | None:
4150 """``"trial_id"`` / ``"text_id"`` when every matched reading took the
4151 same route, ``"mixed"`` when both were used, ``None`` for no match."""
4152 if not self.matched:
4153 return None
4154 if not self.by_text:
4155 return "trial_id"
4156 return "text_id" if not self.by_trial else "mixed"
4158 @property
4159 def label(self) -> str:
4160 """The route as the wizard's fields name it (``"Text ID"``)."""
4161 if self.key == "mixed":
4162 return f"Trial ID ({self.by_trial:,}) and by Text ID ({self.by_text:,})"
4163 return STIMULUS_JOIN_LABELS.get(self.key or "", "")
4165 def _unit(self, n: int) -> str:
4166 return (
4167 _count(n, "trial screen", "trial screens")
4168 if self.multipart
4169 else _count(n, "trial", "trials")
4170 )
4172 def _why_unmatched(self) -> str:
4173 """Why the unmatched readings found nothing, split by the reason."""
4174 parts = []
4175 plain = self.unmatched - self.ambiguous_readings - self.missing_screen_readings
4176 if plain:
4177 verb = "shares" if plain == 1 else "share"
4178 parts.append(
4179 f" {self._unit(plain)} {verb} neither a Trial ID nor a Text ID "
4180 "with the Words table."
4181 )
4182 if self.missing_screen_readings:
4183 screens = ", ".join(repr(s) for s in self.missing_screens[:3])
4184 verb = "matches" if self.missing_screen_readings == 1 else "match"
4185 which = "that screen" if len(self.missing_screens) == 1 else "those screens"
4186 parts.append(
4187 f" {self._unit(self.missing_screen_readings)} {verb} a Words-table trial "
4188 f"that has no boxes for {which} ({screens})."
4189 )
4190 if self.ambiguous_readings:
4191 shown = ", ".join(repr(t) for t in self.ambiguous_texts[:3])
4192 more = len(self.ambiguous_texts) - 3
4193 if more > 0:
4194 shown += f" and {more:,} more"
4195 texts = (
4196 "that Text ID" if len(self.ambiguous_texts) == 1 else "those Text IDs"
4197 )
4198 verb = "names" if self.ambiguous_readings == 1 else "name"
4199 parts.append(
4200 f" {self._unit(self.ambiguous_readings)} {verb} a Text ID the Words "
4201 f"table gives to more than one of its trials ({shown}), so "
4202 f"{texts} cannot pick one set of boxes."
4203 )
4204 return "".join(parts)
4206 def _mismatch_note(self) -> str:
4207 if not self.text_mismatches:
4208 return ""
4209 verb, pron = ("has", "its") if self.text_mismatches == 1 else ("have", "their")
4210 example = ""
4211 if self.mismatch_example is not None:
4212 reading, aoi = self.mismatch_example
4213 example = f" (e.g. {reading!r} against {aoi!r})"
4214 return (
4215 f" {self._unit(self.text_mismatches)} {verb} the boxes of the Words-"
4216 f"table trial {pron} Trial ID matches, but a different Text ID from that "
4217 f"trial's{example}: check that Text ID names the same texts in both "
4218 "tables."
4219 )
4221 def describe(self) -> str:
4222 """One sentence for the wizard and the log: the route and its coverage."""
4223 if self.key is None or (self.multipart and self.unmatched):
4224 return self.problem()
4225 unit = "trial screens" if self.multipart else "trials"
4226 lead = f"Words attach to {unit} by {self.label}"
4227 if not self.unmatched:
4228 return (
4229 f"{lead}: all {self.readings:,} {unit} have word boxes."
4230 f"{self._mismatch_note()}"
4231 )
4232 return (
4233 f"{lead}: {self.matched:,} of {self._unit(self.readings)} have word "
4234 f"boxes.{self._why_unmatched()}{self._mismatch_note()}"
4235 )
4237 def problem(self) -> str:
4238 """Why the join is refused, and what to map instead."""
4239 advice = (
4240 " Map **Text ID** in both tables to the column naming the text "
4241 "each row belongs to, or give the Words table the fixations' own "
4242 "Trial IDs."
4243 )
4244 if self.matched:
4245 # Only a multipart dataset refuses a partial join: every screen a
4246 # reading has fixations on needs its boxes (`validate_matching_parts`).
4247 return (
4248 "The Words table has no Participant ID, so its word boxes are shared "
4249 f"by every trial of a text, but only {self.matched:,} of "
4250 f"{self._unit(self.readings)} find theirs, and a multipart dataset "
4251 "needs boxes for every screen it has fixations on."
4252 f"{self._why_unmatched()}{advice}"
4253 )
4254 return (
4255 "The Words table has no Participant ID, so its word boxes are shared by "
4256 f"every trial of a text, but none of the {self._unit(self.readings)} "
4257 "in the fixations finds them: the dataset would have no word boxes."
4258 f"{self._why_unmatched()}{advice}"
4259 )
4262def _as_key(values: pd.Series) -> pd.Series:
4263 """An id column as the strings the join compares, missing ids kept missing
4264 (``astype(str)`` alone would spell them ``"nan"`` and match each other)."""
4265 return values.astype(str).where(values.notna())
4268def _text_is_fallback(frame: pd.DataFrame) -> bool:
4269 """Whether ``text_id`` is only the trial-id fallback on this frame.
4271 Read from ``_text_id_mapped``, which ``normalize_*`` write from the schema,
4272 so a mapped Text ID is mapped even when its values equal the trial ids.
4273 Only a frame normalized without it (a stored dataset from before, frames
4274 built by hand) falls back to the values: with no Text ID mapped,
4275 ``normalize_*`` copy the (unsuffixed) trial id into ``text_id``, so a
4276 ``text_id`` equal to that id on every row says nothing about texts.
4277 Compared over distinct pairs, so it is cheap on a million rows."""
4278 if TEXT_ID_MAPPED in frame.columns:
4279 return not bool(frame[TEXT_ID_MAPPED].fillna(False).astype(bool).any())
4280 columns = ["text_id", "trial_id"] + [BASE_TRIAL_ID] * (BASE_TRIAL_ID in frame)
4281 pairs = frame[columns].drop_duplicates()
4282 trial = pairs["trial_id"]
4283 if BASE_TRIAL_ID in pairs:
4284 trial = pairs[BASE_TRIAL_ID].where(pairs[BASE_TRIAL_ID].notna(), trial)
4285 return bool(
4286 (
4287 _as_key(pairs["text_id"]).fillna("\x00") == _as_key(trial).fillna("\x00")
4288 ).all()
4289 )
4292def _lookup(
4293 readings: pd.DataFrame,
4294 keys: pd.Series,
4295 table: pd.DataFrame,
4296 screen: list,
4297 value: str = _STIMULUS_KEY,
4298) -> np.ndarray:
4299 """For each reading, ``table[value]`` where ``table["_key"]`` equals ``keys``.
4301 ``table`` is unique on ``_key`` + ``screen``; a left merge keeps the
4302 readings' order, so the result lines up with them (missing = no match).
4303 A blank screen id pairs with a blank one, as the merge always has."""
4304 probe = pd.DataFrame(
4305 {"_key": keys.to_numpy(), **{c: readings[c].to_numpy() for c in screen}}
4306 )
4307 hits = probe.merge(
4308 table[["_key", *screen, value]], on=["_key", *screen], how="left"
4309 )
4310 return hits[value].to_numpy()
4313def _plan_stimulus_join(
4314 words: pd.DataFrame, fixations: pd.DataFrame
4315) -> tuple[StimulusJoin, pd.DataFrame, list[str]]:
4316 """The join, the fixations' distinct readings, and the reading key.
4318 Each reading's ``_STIMULUS_KEY`` is the AOI table trial whose boxes it
4319 takes (missing when none): its exact trial id, else its unsuffixed one,
4320 else — only when the fixations map a real Text ID — its Text ID's one
4321 trial. Vectorised over *distinct* keys (one ``drop_duplicates`` per frame
4322 and one merge per route), so it costs the same on a million-row corpus as
4323 the broadcast."""
4324 screen = [SCREEN_ID] * (
4325 SCREEN_ID in words.columns and SCREEN_ID in fixations.columns
4326 )
4327 reading_key = ["participant_id", "trial_id", *screen]
4328 has_text = "text_id" in words.columns and "text_id" in fixations.columns
4329 # A fixations Text ID that is only its own trial id names no text, so it
4330 # can neither find a text's boxes nor contradict them; the AOI table's
4331 # fallback text *is* its stimulus id, which a real reading Text ID can use.
4332 reading_text_real = has_text and not _text_is_fallback(fixations)
4333 aoi_text_real = has_text and not _text_is_fallback(words)
4334 extra = ["text_id"] * has_text + [BASE_TRIAL_ID] * (
4335 BASE_TRIAL_ID in fixations.columns
4336 )
4337 readings = fixations[reading_key + extra].drop_duplicates(reading_key)
4338 readings = readings.reset_index(drop=True)
4339 trials = words[["trial_id", *screen, *["text_id"] * has_text]]
4340 trials = trials.dropna(subset=["trial_id"]).drop_duplicates(["trial_id", *screen])
4341 trials = trials.assign(_key=_as_key(trials["trial_id"]))
4342 trials[_STIMULUS_KEY] = trials["_key"]
4344 def by_trial_routes(table: pd.DataFrame, on_screen: list) -> np.ndarray:
4345 found = _lookup(readings, _as_key(readings["trial_id"]), table, on_screen)
4346 if BASE_TRIAL_ID in readings.columns:
4347 # A repeat's `_r2` tells the readings apart; it never costs the
4348 # repeat its boxes (BUG-57): try the id it was recorded under.
4349 base = _lookup(readings, _as_key(readings[BASE_TRIAL_ID]), table, on_screen)
4350 found = np.where(pd.isna(found), base, found)
4351 return found
4353 stimulus = by_trial_routes(trials, screen)
4354 by_trial = int(pd.notna(stimulus).sum())
4356 ambiguous: tuple[str, ...] = ()
4357 ambiguous_readings = mismatches = 0
4358 mismatch_example = None
4359 texts = None
4360 if has_text:
4361 reading_text = _as_key(readings["text_id"])
4362 if reading_text_real and aoi_text_real:
4363 # An exact trial-id match always stands (a Text ID mapped at another
4364 # grain, or on one side only, must never move a reading's boxes);
4365 # when both tables name real texts and they disagree, say so.
4366 matched_text = _lookup(
4367 readings,
4368 pd.Series(stimulus),
4369 trials.assign(_text=_as_key(trials["text_id"])),
4370 screen,
4371 value="_text",
4372 )
4373 disagree = (
4374 pd.notna(stimulus)
4375 & pd.notna(matched_text)
4376 & reading_text.notna().to_numpy()
4377 & (matched_text != reading_text.to_numpy())
4378 )
4379 mismatches = int(disagree.sum())
4380 if mismatches:
4381 i = int(np.flatnonzero(disagree)[0])
4382 mismatch_example = (str(reading_text.iloc[i]), str(matched_text[i]))
4383 if reading_text_real:
4384 text_key = ["text_id", *screen]
4385 # A Text ID stands in for the trial only where it names one set of
4386 # boxes: one the AOI table gives to several of its trials (two
4387 # versions of a paragraph, an article id over paragraph boxes) would
4388 # hand a reading all of them. Only those texts are left out.
4389 per_text = words[[*text_key, "trial_id"]]
4390 per_text = per_text.dropna(subset=["text_id", "trial_id"]).drop_duplicates()
4391 shared = per_text.duplicated(text_key, keep=False)
4392 ambiguous = tuple(
4393 sorted(per_text.loc[shared, "text_id"].astype(str).unique())
4394 )
4395 chosen = per_text[~shared]
4396 texts = pd.DataFrame(
4397 {
4398 "_key": _as_key(chosen["text_id"]).to_numpy(),
4399 **{c: chosen[c].to_numpy() for c in screen},
4400 _STIMULUS_KEY: _as_key(chosen["trial_id"]).to_numpy(),
4401 }
4402 )
4403 by_text_id = _lookup(readings, reading_text, texts, screen)
4404 stimulus = np.where(pd.isna(stimulus), by_text_id, stimulus)
4405 if ambiguous:
4406 ambiguous_readings = int(
4407 (pd.isna(stimulus) & reading_text.isin(ambiguous).to_numpy()).sum()
4408 )
4409 matched = int(pd.notna(stimulus).sum())
4411 missing_screen_readings = 0
4412 missing_screens: tuple[str, ...] = ()
4413 if screen and matched < len(readings):
4414 # Matched an AOI trial, just not on this screen: say which screen.
4415 anywhere = by_trial_routes(trials.drop_duplicates(["trial_id"]), [])
4416 if texts is not None:
4417 anywhere = np.where(
4418 pd.isna(anywhere),
4419 _lookup(
4420 readings,
4421 _as_key(readings["text_id"]),
4422 texts.drop_duplicates(["_key"]),
4423 [],
4424 ),
4425 anywhere,
4426 )
4427 missing = pd.isna(stimulus) & pd.notna(anywhere)
4428 missing_screen_readings = int(missing.sum())
4429 missing_screens = tuple(
4430 dict.fromkeys(str(s) for s in readings.loc[missing, SCREEN_ID])
4431 )
4432 if ambiguous_readings:
4433 ambiguous_readings = int(
4434 (
4435 pd.isna(stimulus)
4436 & ~missing
4437 & _as_key(readings["text_id"]).isin(ambiguous).to_numpy()
4438 ).sum()
4439 )
4440 join = StimulusJoin(
4441 readings=len(readings),
4442 by_trial=by_trial,
4443 by_text=matched - by_trial,
4444 ambiguous_texts=ambiguous,
4445 ambiguous_readings=ambiguous_readings,
4446 missing_screen_readings=missing_screen_readings,
4447 missing_screens=missing_screens,
4448 text_mismatches=mismatches,
4449 mismatch_example=mismatch_example,
4450 multipart=bool(screen),
4451 )
4452 readings[_STIMULUS_KEY] = stimulus
4453 return join, readings, reading_key
4456def plan_stimulus_join(
4457 words: pd.DataFrame, fixations: pd.DataFrame
4458) -> StimulusJoin | None:
4459 """How :func:`broadcast_stimulus_words` attaches ``words`` (DATA-49).
4461 ``None`` when there is nothing to join: the words carry a Participant ID
4462 (so they are not stimulus-level), or either frame is empty. Otherwise the
4463 :class:`StimulusJoin` it would make — ``key=None`` when it would refuse.
4464 ``harmonize_frames`` runs other fixups first (BUG-59's zero padding), so
4465 for the join a load actually made, use :func:`harmonize_frames_with_join`.
4466 """
4467 if STIMULUS_WORDS_FLAG not in words.columns or words.empty or fixations.empty:
4468 return None
4469 return _plan_stimulus_join(words, fixations)[0]
4472def _broadcast_stimulus_words(
4473 words: pd.DataFrame, fixations: pd.DataFrame
4474) -> tuple[pd.DataFrame, StimulusJoin | None]:
4475 """:func:`broadcast_stimulus_words`, plus the join it made (``None`` when
4476 it made none: per-reader words, or nothing to broadcast against)."""
4477 if STIMULUS_WORDS_FLAG not in words.columns:
4478 return words, None
4479 if words.empty or fixations.empty:
4480 words = words.drop(columns=[STIMULUS_WORDS_FLAG])
4481 # No fixations to broadcast across (e.g. a words-only dataset): there's a
4482 # single anonymous reader, so give the placeholder a real synthetic id.
4483 if not words.empty:
4484 words = words.copy()
4485 words["participant_id"] = SYNTHETIC_PARTICIPANT
4486 return words, None
4487 join, readings, reading_key = _plan_stimulus_join(words, fixations)
4488 # A multipart dataset needs boxes for every screen it has fixations on
4489 # (`validate_matching_parts` would reject it a step later as "orphan
4490 # screens"), so a partial join is refused here, with the reasons.
4491 if join.key is None or (join.multipart and join.unmatched):
4492 raise StimulusJoinError(join.problem())
4493 if join.needs_warning:
4494 # The API and the CLI have no page to put this on, and a script would
4495 # otherwise get readings with no boxes, or boxes whose Text ID
4496 # disagrees with the reading's, without a word.
4497 warnings.warn(join.describe(), StimulusJoinWarning, stacklevel=4)
4498 else:
4499 _LOGGER.info(join.describe())
4500 screen = reading_key[2:]
4501 on = [_STIMULUS_KEY, *screen]
4502 # The AOI table supplies the boxes (and its own Text ID); the reading
4503 # supplies its identity — the AOI table's trial id is the stimulus', not
4504 # the reading's, and it stays on each copy as `_aoi_trial_id`.
4505 stimulus = words.drop(
4506 columns=[c for c in (STIMULUS_WORDS_FLAG, AOI_TRIAL_ID) if c in words]
4507 + ["participant_id"]
4508 ).rename(columns={"trial_id": _STIMULUS_KEY})
4509 # Only the key must be present: a blank screen pairs with a blank screen.
4510 stimulus = stimulus.dropna(subset=[_STIMULUS_KEY])
4511 stimulus[_STIMULUS_KEY] = stimulus[_STIMULUS_KEY].astype(str)
4512 pairs = readings[[*reading_key, _STIMULUS_KEY]].dropna(subset=[_STIMULUS_KEY])
4513 pairs = pairs.assign(
4514 participant_id=pairs["participant_id"].astype(str),
4515 trial_id=pairs["trial_id"].astype(str),
4516 )
4517 out = stimulus.merge(pairs, on=on, how="inner").rename(
4518 columns={_STIMULUS_KEY: AOI_TRIAL_ID}
4519 )
4520 if "unique_trial_id" in out.columns:
4521 # The reading's id *is* its unique trial id (BUG-58).
4522 out["unique_trial_id"] = out["trial_id"]
4523 return out, join
4526def broadcast_stimulus_words(
4527 words: pd.DataFrame, fixations: pd.DataFrame
4528) -> pd.DataFrame:
4529 """Give every reading its own copy of the stimulus-level boxes of its text.
4531 Datasets like PoTeC ship word/AoI tables per *text* (no participant
4532 column) while fixations are per participant × text. After normalization,
4533 such words carry the ``_stimulus_words`` flag; this gives each reading —
4534 each ``(participant_id, trial_id[, screen_id])`` of the fixations — the
4535 boxes of the AOI table's trial it belongs to, stamped with that reading's
4536 own ids (the AOI trial it came from kept as ``_aoi_trial_id``), so
4537 downstream (participant, trial) filtering works unchanged. Words for texts
4538 nobody read are dropped.
4540 DATA-49: one rule, per reading, whatever the trial ids look like. A reading
4541 takes the boxes of the first AOI trial it finds by its own trial id (ids
4542 shared across readers), by the id it had before a repeat's ``_r2`` suffix
4543 (``_base_trial_id``, BUG-57), or by its **Text ID** (trial ids that embed
4544 the reader). The Text-ID route needs a real fixations Text ID (mapped, not
4545 the trial-id fallback — ``_text_id_mapped``), and never uses one the AOI
4546 table gives to several trials. A trial-id match always stands; mapped Text
4547 IDs that disagree with it are counted and warned about, never redirected.
4548 When no reading finds any — or, on a multipart dataset, when any
4549 reading screen finds none — it raises :class:`StimulusJoinError` rather than
4550 return a dataset without them, and when only some do it warns
4551 (:class:`StimulusJoinWarning`). One vectorised merge — it runs on
4552 million-row corpora.
4554 No-op for ordinary per-participant word tables. With no fixations to
4555 broadcast across (a words-only dataset) the rows get the synthetic reader."""
4556 return _broadcast_stimulus_words(words, fixations)[0]
4559def repair_stranded_stimulus_words(
4560 words: pd.DataFrame, fixations: pd.DataFrame
4561) -> tuple[pd.DataFrame, pd.DataFrame] | None:
4562 """DATA-39 — re-broadcast a stored AOI table the old ✅ Save changes stranded.
4564 Before DATA-39 was fixed, saving an edit to a dataset whose AOI table has no
4565 participant column left every word on the ``""`` placeholder reader with
4566 the ``_stimulus_words`` flag still set, so no trial found its boxes. A
4567 *stored* frame can only carry that flag through that bug —
4568 ``broadcast_stimulus_words`` always drops it — so its presence is the
4569 diagnosis, and running the broadcast it missed is the repair. Returns the
4570 repaired ``(words, fixations)``, or ``None`` when there is nothing to repair
4571 or the frames will not harmonize (the dataset is then left as it was).
4572 """
4573 if not isinstance(words, pd.DataFrame) or STIMULUS_WORDS_FLAG not in words.columns:
4574 return None
4575 has_fixations = isinstance(fixations, pd.DataFrame) and not fixations.empty
4576 try:
4577 repaired, harmonized = harmonize_frames(
4578 words, fixations if has_fixations else empty_fixations_frame()
4579 )
4580 except Exception: # a repair must never break the load
4581 _LOGGER.warning(
4582 "Could not repair a stored Words table left on the placeholder "
4583 "participant; press Save changes on the Edit dataset screen to retry.",
4584 exc_info=True,
4585 )
4586 return None
4587 if repaired.empty:
4588 return None
4589 return repaired, (harmonized if has_fixations else fixations)
4592def fill_fixation_xy_from_words(
4593 fixations: pd.DataFrame, words: pd.DataFrame
4594) -> pd.DataFrame:
4595 """Fill missing fixation coordinates from the fixated word's box center.
4597 AOI-sequence datasets record *which* word/character each fixation landed
4598 on but not the pixel position. When normalized fixations have NaN x/y and
4599 a ``word_id``, place them at the center of the matching word box (keyed by
4600 participant_id + trial_id + word_id). Fixations whose word_id matches no
4601 box keep NaN coordinates. Rows that already have coordinates are left
4602 untouched."""
4603 if fixations.empty or words.empty:
4604 return fixations
4605 missing = fixations["x"].isna() | fixations["y"].isna()
4606 if not missing.any() or "word_id" not in fixations.columns:
4607 return fixations
4608 from .measures import word_box_bounds
4610 # The interest area's own centre (BUG-83), which is inside the box the
4611 # assignment will then test it against.
4612 x0, y0, x1, y1 = word_box_bounds(words)
4613 keys = grouping_columns(words, include_word=True)
4614 centers = words[keys].copy()
4615 centers["_word_cx"] = (x0 + x1) / 2.0
4616 centers["_word_cy"] = (y0 + y1) / 2.0
4617 centers = centers.drop_duplicates(keys)
4618 merged = fixations[keys].merge(centers, on=keys, how="left")
4619 fixations = fixations.copy()
4620 fill = missing.to_numpy()
4621 fixations.loc[fill, "x"] = merged["_word_cx"].to_numpy()[fill]
4622 fixations.loc[fill, "y"] = merged["_word_cy"].to_numpy()[fill]
4623 return fixations
4626def _reconcile_participant_asymmetry(
4627 words: pd.DataFrame, fixations: pd.DataFrame
4628) -> pd.DataFrame:
4629 """Re-key word boxes to the synthetic participant when the fixations have no
4630 participant but the words do.
4632 With participant now optional per table, a fixations table can be
4633 participant-less (every row stamped ``SYNTHETIC_PARTICIPANT``) while the words
4634 table still carries real participant ids. The trial picker keys off the
4635 fixations, so it offers ``(all)`` — but the boxes are keyed by the real ids
4636 and ``extract_trial`` then finds none, rendering fixations with no text. Stamp
4637 the words with the synthetic id (dropping the now-duplicate per-reader boxes)
4638 so they line up. No-op unless the fixations are entirely synthetic and the
4639 words are not — the stimulus-words broadcast already covers the reverse."""
4640 if words.empty or fixations.empty or "participant_id" not in words.columns:
4641 return words
4642 if set(fixations["participant_id"].unique()) != {SYNTHETIC_PARTICIPANT}:
4643 return words
4644 word_parts = set(words["participant_id"].unique())
4645 if not word_parts or word_parts == {SYNTHETIC_PARTICIPANT}:
4646 return words
4647 words = words.copy()
4648 words["participant_id"] = SYNTHETIC_PARTICIPANT
4649 subset = grouping_columns(words, include_word=True)
4650 if subset:
4651 words = words.drop_duplicates(subset=subset)
4652 return words
4655_WORD_ID_AGGS = ["min", "max", "nunique"]
4658def _key_frame(frame: pd.DataFrame, ids: pd.Series) -> pd.DataFrame:
4659 """``(participant_id, trial_id, _id)`` view of ``frame``, aligned to ``ids``.
4661 ``ids`` is a NaN-dropped numeric word-id series taken from ``frame``, so the
4662 id columns are re-indexed onto its (subset) index.
4663 """
4664 data = {
4665 column: frame[column].reindex(ids.index) for column in grouping_columns(frame)
4666 }
4667 data["_id"] = ids
4668 return pd.DataFrame(data)
4671def detect_word_id_offset(words: pd.DataFrame, fixations: pd.DataFrame) -> int:
4672 """Detect a 1-based fixation ``word_id`` against 0-based word boxes (BUG-8).
4674 Some exports (the bundled OneStop demo among them) number the fixation
4675 report's word column ``1..N`` while the interest-area table numbers its rows
4676 ``0..N-1``, so every fixation's pre-assigned ``word_id`` points at the *next*
4677 word. ``measures.assign_fixations_to_words`` keeps existing ids, so the
4678 computed reading measures then attach to the wrong words.
4680 Returns the offset to **subtract** from the fixation word ids: ``1`` when the
4681 shift is unambiguous, ``0`` (by far the common case) otherwise. A false
4682 positive silently corrupts a correct dataset, so the test is deliberately
4683 strict — every condition below must hold across the whole dataset:
4685 * both frames carry whole-number ``word_id`` values,
4686 * the words ids start at ``0`` and the fixation ids start at ``1``,
4687 * *every* trial present in both frames has 0-based, gap-free word ids and no
4688 fixation id below ``1`` or more than one past its last word, and
4689 * at least one trial actually overflows by exactly one
4690 (``max fixation id == max word id + 1``) — without that there is no
4691 evidence of a shift, just a reader who never looked at the first word.
4692 """
4693 if words is None or fixations is None or words.empty or fixations.empty:
4694 return 0
4695 keys = grouping_columns(words)
4696 if keys != grouping_columns(fixations):
4697 return 0
4698 needed = set(keys) | {"word_id"}
4699 if not needed.issubset(words.columns) or not needed.issubset(fixations.columns):
4700 return 0
4701 w_ids = pd.to_numeric(words["word_id"], errors="coerce").dropna()
4702 f_ids = pd.to_numeric(fixations["word_id"], errors="coerce").dropna()
4703 if w_ids.empty or f_ids.empty:
4704 return 0
4705 # Fractional ids (character-level indices, say) aren't a word numbering we
4706 # can reason about.
4707 if not (w_ids % 1 == 0).all() or not (f_ids % 1 == 0).all():
4708 return 0
4709 if float(w_ids.min()) != 0.0 or float(f_ids.min()) != 1.0:
4710 return 0
4711 # Project onto (keys + id) rather than .assign()-ing onto the source frames:
4712 # the OneStop words/fixations tables are wide, and this runs on every load.
4713 w_stats = (
4714 _key_frame(words, w_ids).groupby(keys, sort=False)["_id"].agg(_WORD_ID_AGGS)
4715 )
4716 f_stats = (
4717 _key_frame(fixations, f_ids)
4718 .groupby(keys, sort=False)["_id"]
4719 .agg(["min", "max"])
4720 )
4721 joined = w_stats.join(f_stats, how="inner", lsuffix="_w", rsuffix="_f")
4722 if joined.empty:
4723 return 0
4724 zero_based = joined["min_w"] == 0
4725 gap_free = joined["nunique"] == joined["max_w"] + 1
4726 in_range = joined["min_f"] >= 1
4727 overflow = joined["max_f"] == joined["max_w"] + 1
4728 runaway = joined["max_f"] > joined["max_w"] + 1
4729 if not (zero_based.all() and gap_free.all() and in_range.all()):
4730 return 0
4731 if runaway.any() or not overflow.any():
4732 return 0
4733 return 1
4736def correct_word_id_offset(
4737 words: pd.DataFrame, fixations: pd.DataFrame, *, offset: int | None = None
4738) -> pd.DataFrame:
4739 """Shift fixation ``word_id`` back onto the words table when it's 1-based.
4741 No-op unless :func:`detect_word_id_offset` finds an unambiguous shift.
4742 Renumbering someone's ids is never silent — it's logged at INFO, which
4743 `debug_log.install_log_capture` surfaces in the in-app 🐛 Debug panel. Not a
4744 `st.warning`, and not a WARNING either (BUG-76): the bundled demo corpus
4745 trips this on *every* load, so a banner would be permanent furniture on the
4746 landing view, and a WARNING was the first line `render --sample`,
4747 `load_sample_data()` and the README quickstart printed to a new user's
4748 terminal — about a correction that needs nothing from them.
4750 ``offset`` is :func:`detect_word_id_offset`'s answer when the caller
4751 already asked it (DATA-66's harmonize report).
4752 """
4753 if offset is None:
4754 offset = detect_word_id_offset(words, fixations)
4755 if not offset:
4756 return fixations
4757 fixations = fixations.copy()
4758 fixations["word_id"] = pd.to_numeric(fixations["word_id"], errors="coerce") - offset
4759 _LOGGER.info(
4760 "The fixation report's word ids are numbered from 1 while the word "
4761 "boxes are numbered from 0, so every fixation pointed at the next word. "
4762 "Shifted the fixation word ids down by %d to line the two tables up.",
4763 offset,
4764 )
4765 return fixations
4768def _pad_ids(frame: pd.DataFrame, column: str, mapping: dict) -> pd.DataFrame:
4769 """``frame`` with ``column`` respelled by a zero-padding ``mapping``.
4771 Padding a trial id also pads what was copied from it: a repeat's
4772 ``_base_trial_id``, and a fallback ``text_id`` (no Text ID mapped, so it
4773 *is* the trial id) — left at "7" beside a trial "007" it would read as a
4774 real Text ID of its own (DATA-49)."""
4775 frame = frame.copy()
4776 if column == "trial_id":
4777 if "text_id" in frame.columns:
4778 copied = _as_key(frame["text_id"]) == _as_key(frame["trial_id"])
4779 if TEXT_ID_MAPPED in frame.columns:
4780 copied &= ~frame[TEXT_ID_MAPPED].fillna(False).astype(bool)
4781 frame.loc[copied, "text_id"] = frame.loc[copied, "text_id"].replace(mapping)
4782 if BASE_TRIAL_ID in frame.columns:
4783 # A repeat is joined by the id it was suffixed from.
4784 frame[BASE_TRIAL_ID] = frame[BASE_TRIAL_ID].replace(mapping)
4785 frame[column] = frame[column].replace(mapping)
4786 return frame
4789def _restore_zero_padding(
4790 words: pd.DataFrame,
4791 fixations: pd.DataFrame,
4792 padded: list[tuple[str, str]] | None = None,
4793) -> tuple[pd.DataFrame, pd.DataFrame]:
4794 """Spell a zero-padded id the same way in both frames (BUG-59).
4796 A CSV read ``007`` as 7 while a Parquet table kept "007", and the two
4797 tables then shared no participant — every fixation drew over no text. When
4798 padding is the only difference (:func:`zero_padding_map`), the side that
4799 lost its zeros is given them back, and the rename is logged — and recorded
4800 in ``padded`` as ``(table, column)`` when given.
4801 """
4802 if words.empty or fixations.empty:
4803 return words, fixations
4804 columns = ["trial_id", "text_id"]
4805 if STIMULUS_WORDS_FLAG not in words.columns:
4806 columns.insert(0, "participant_id")
4807 # The second respelling: one table stored before composite ids escaped a
4808 # `_` inside a part (`compose_id`), the other composed since — a table
4809 # added on ✏️ Edit dataset to a dataset restored from the recovery cache.
4810 respellings = (
4811 (zero_padding_map, "without the zero-padding the other table uses"),
4812 (composite_respelling_map, "with the other spelling of a composite id"),
4813 )
4814 for column in columns:
4815 if column not in words.columns or column not in fixations.columns:
4816 continue
4817 w_ids, f_ids = words[column].unique(), fixations[column].unique()
4818 for respell, how in respellings:
4819 for frame_name, ids, reference in (
4820 ("words", w_ids, f_ids),
4821 ("fixations", f_ids, w_ids),
4822 ):
4823 mapping = respell(ids, reference)
4824 if not mapping:
4825 continue
4826 if frame_name == "words":
4827 words = _pad_ids(words, column, mapping)
4828 else:
4829 fixations = _pad_ids(fixations, column, mapping)
4830 if padded is not None:
4831 padded.append((frame_name, column))
4832 _LOGGER.info(
4833 "The %s table spelled %d %s value(s) %s (e.g. %r for %r); "
4834 "matched them up.",
4835 frame_name,
4836 len(mapping),
4837 column,
4838 how,
4839 *next(iter(mapping.items()))[::-1],
4840 )
4841 w_ids, f_ids = words[column].unique(), fixations[column].unique()
4842 break
4843 return words, fixations
4846def harmonize_frames_with_join(
4847 words: pd.DataFrame, fixations: pd.DataFrame
4848) -> tuple[pd.DataFrame, pd.DataFrame, StimulusJoin | None]:
4849 """Cross-frame fixups applied right after normalization.
4851 Match zero-padded ids the two tables spell differently (BUG-59), broadcast
4852 stimulus-level words across participants, reconcile a participant-less
4853 fixations table with participant-bearing words, correct a 1-based fixation
4854 ``word_id`` (BUG-8), then fill missing fixation coordinates from word-box
4855 centers. Call whenever both frames are available (the API and the app both
4856 route through this). Also returns the :class:`StimulusJoin` a
4857 stimulus-level AOI table was attached by — ``None`` for a per-reader one —
4858 which the add-dataset wizard states (DATA-49)."""
4859 words, fixations, join, _rewrites = harmonize_frames_reporting(words, fixations)
4860 return words, fixations, join
4863#: DATA-66: a column whose values normalization or the fixups changed, as
4864#: ``(table, column, how)`` — ``how`` follows the source column's name in its
4865#: label ("CURRENT_FIX_INTEREST_AREA_ID − 1"), since a header the file used must
4866#: not sit over values the file never held. ``column_names.with_rewrites``
4867#: applies them.
4868Rewrite = tuple[str, str, str]
4871def harmonize_frames_reporting(
4872 words: pd.DataFrame, fixations: pd.DataFrame
4873) -> tuple[pd.DataFrame, pd.DataFrame, StimulusJoin | None, tuple[Rewrite, ...]]:
4874 """:func:`harmonize_frames_with_join`, also saying which columns' values it
4875 changed (:data:`Rewrite`): ids it zero-padded (BUG-59), a fixation
4876 ``word_id`` shifted onto 0-based boxes (BUG-8), fixation positions filled
4877 from word boxes, and a repeated reading's ``_rN`` trial id (BUG-57)."""
4878 from .preprocessing import add_text_direction
4880 rewrites: list[Rewrite] = []
4881 padded: list[tuple[str, str]] = []
4882 words, fixations = _restore_zero_padding(words, fixations, padded)
4883 rewrites += [
4884 (table, column, ", respelled to match the other table")
4885 for table, column in padded
4886 ]
4887 words, join = _broadcast_stimulus_words(words, fixations)
4888 words = add_text_direction(words)
4889 words = _reconcile_participant_asymmetry(words, fixations)
4890 words = normalize_screen_identity(words)
4891 fixations = normalize_screen_identity(fixations)
4892 validate_matching_parts(words, fixations)
4893 offset = detect_word_id_offset(words, fixations)
4894 fixations = correct_word_id_offset(words, fixations, offset=offset)
4895 if offset:
4896 rewrites.append(("fixations", "word_id", f" − {offset}"))
4897 blank = (
4898 fixations[["x", "y"]].isna().sum()
4899 if {"x", "y"} <= set(fixations.columns)
4900 else None
4901 )
4902 fixations = fill_fixation_xy_from_words(fixations, words)
4903 if blank is not None:
4904 filled = blank - fixations[["x", "y"]].isna().sum()
4905 rewrites += [
4906 ("fixations", axis, ", blanks filled from the word boxes")
4907 for axis in ("x", "y")
4908 if filled[axis] > 0
4909 ]
4910 if (
4911 BASE_TRIAL_ID in fixations.columns
4912 and (_as_key(fixations["trial_id"]) != _as_key(fixations[BASE_TRIAL_ID]))
4913 .where(fixations[BASE_TRIAL_ID].notna(), False)
4914 .any()
4915 ):
4916 rewrites.append(("fixations", "trial_id", " + _rN for a repeated trial"))
4917 return words, fixations, join, tuple(rewrites)
4920def harmonize_frames(
4921 words: pd.DataFrame, fixations: pd.DataFrame
4922) -> tuple[pd.DataFrame, pd.DataFrame]:
4923 """:func:`harmonize_frames_with_join` without the join report."""
4924 words, fixations, _join = harmonize_frames_with_join(words, fixations)
4925 return words, fixations
4928def _disambiguate_repeated_readings(
4929 df: pd.DataFrame,
4930 source: pd.DataFrame,
4931 trial_col: str,
4932 *,
4933 record_base: bool = False,
4934) -> pd.DataFrame:
4935 """Suffix `trial_id` with `_r2`, `_r3` … when a participant read the same
4936 paragraph more than once.
4938 OneStop L2's per-pid parquet shards don't carry a `unique_trial_id` column,
4939 so the schema-inference fallback uses `unique_paragraph_id` — but that's
4940 the same string for both readings of a repeated-reading trial. Without
4941 this fix, the two readings' fixations collapse into one scanpath (and into
4942 one row of the trial picker), which is what the cached PNG thumbnails
4943 (which filter on TRIAL_INDEX) correctly avoid. We rank by TRIAL_INDEX so
4944 the chronologically-first reading keeps its original id; later readings
4945 get `_r2`, `_r3`, … appended.
4947 Groups on the already-computed ``df["participant_id"]`` (1:1 with ``source``),
4948 so a composite participant id is handled without recomputing the join.
4949 """
4950 if trial_col == "unique_trial_id":
4951 return df
4952 idx_col = next(
4953 (c for c in ("TRIAL_INDEX", "trial_index") if c in source.columns), None
4954 )
4955 if idx_col is None:
4956 return df
4957 grouper = pd.DataFrame(
4958 {
4959 "_pk": df["participant_id"].to_numpy(),
4960 "_tc": source[trial_col].astype(str).to_numpy(),
4961 "_idx": source[idx_col].to_numpy(),
4962 }
4963 )
4964 # A reading with no index of its own keeps its id unsuffixed rather than
4965 # crashing the cast (BUG-56).
4966 rank = (
4967 grouper.groupby(["_pk", "_tc"])["_idx"]
4968 .rank(method="dense")
4969 .fillna(1)
4970 .astype(int)
4971 .to_numpy()
4972 )
4973 base = df["trial_id"]
4974 df["trial_id"] = [
4975 tid if r == 1 else f"{tid}_r{r}" for tid, r in zip(base.to_numpy(), rank)
4976 ]
4977 if record_base and (rank > 1).any():
4978 # What the reading was recorded under, so a table keyed by the stimulus
4979 # still finds a repeat's boxes (BUG-57 / DATA-49's trial join).
4980 df[BASE_TRIAL_ID] = base
4981 return df
4984def has_explicit_trial_index(frame: pd.DataFrame) -> bool:
4985 """True when the data already carries a per-trial index column."""
4986 return any(c in frame.columns for c in ("trial_index", "TRIAL_INDEX"))
4989def trial_order_label(frame: pd.DataFrame) -> str:
4990 """The axis title for :func:`derive_trial_index`'s values: the column it
4991 reads, or how it ordered the trials without one (#374 F35 — the demo
4992 carries both ``trial_index`` and ``TRIAL_INDEX``, which differ)."""
4993 for col in ("trial_index", "TRIAL_INDEX"):
4994 if col in frame.columns:
4995 return f"Trial order ({col})"
4996 if "timestamp_ms" in frame.columns:
4997 return "Trial order (by fixation time)"
4998 return "Trial order (as listed)"
5001def derive_trial_index(frame: pd.DataFrame) -> pd.Series:
5002 """Per-participant 1-based trial order, aligned to ``frame``'s rows.
5004 Prefers an existing ``trial_index`` / ``TRIAL_INDEX`` column (the order the
5005 data already records). Otherwise it ranks each participant's trials by their
5006 earliest ``timestamp_ms`` (falling back to first-appearance order when no
5007 timestamps) and numbers them 1, 2, 3, …. Used by the Corpus Analysis tab to
5008 plot a metric as a function of where the trial fell in the session. Returns a
5009 float Series (NaN where the index can't be determined)."""
5010 if frame.empty or not {"participant_id", "trial_id"} <= set(frame.columns):
5011 return pd.Series([np.nan] * len(frame), index=frame.index, dtype="float64")
5012 for col in ("trial_index", "TRIAL_INDEX"):
5013 if col in frame.columns:
5014 return pd.to_numeric(frame[col], errors="coerce")
5015 if "timestamp_ms" in frame.columns:
5016 order_key = frame.groupby(["participant_id", "trial_id"])[
5017 "timestamp_ms"
5018 ].transform("min")
5019 else:
5020 # First-appearance order: row position of each trial's first row.
5021 order_key = pd.Series(range(len(frame)), index=frame.index)
5022 order_key = (
5023 frame.assign(_k=order_key)
5024 .groupby(["participant_id", "trial_id"])["_k"]
5025 .transform("min")
5026 )
5027 per_trial = (
5028 frame[["participant_id", "trial_id"]]
5029 .assign(_k=pd.to_numeric(order_key, errors="coerce"))
5030 .drop_duplicates(["participant_id", "trial_id"])
5031 .sort_values(["participant_id", "_k"])
5032 )
5033 per_trial["_idx"] = per_trial.groupby("participant_id").cumcount() + 1
5034 merged = frame[["participant_id", "trial_id"]].merge(
5035 per_trial[["participant_id", "trial_id", "_idx"]],
5036 on=["participant_id", "trial_id"],
5037 how="left",
5038 )
5039 return pd.Series(
5040 pd.to_numeric(merged["_idx"], errors="coerce").to_numpy(),
5041 index=frame.index,
5042 dtype="float64",
5043 )
5046# ---------------------------------------------------------------------------
5047# Optional-field registry. Drives (a) which known optional source columns are
5048# carried into the normalized frame and (b) the setup wizard's opt-out checklist.
5049# Each entry: (source, dest, kind, category) where `kind` ∈
5050# {numeric, string, boolean, passthrough} and `category` ∈
5051# {measure, linguistic, meta} groups the fields in the UI. Matched by exact
5052# source name (same as the legacy keep-lists this replaced).
5053# ---------------------------------------------------------------------------
5054WORD_OPTIONAL_FIELDS = [
5055 ("IA_FIRST_FIXATION_DURATION", "first_fixation_ms", "numeric", "measure"),
5056 ("IA_DWELL_TIME", "total_fixation_duration_ms", "numeric", "measure"),
5057 ("IA_FIRST_RUN_DWELL_TIME", "first_pass_gaze_duration_ms", "numeric", "measure"),
5058 (
5059 "IA_SECOND_RUN_DWELL_TIME",
5060 "second_pass_duration_ms",
5061 "numeric",
5062 "measure",
5063 ),
5064 # Compatibility aliases retained for existing datasets/API consumers; the
5065 # PRE-4 canonical field above is the one measure computation consults.
5066 (
5067 "IA_SECOND_RUN_DWELL_TIME",
5068 "higher_pass_fixation_duration_ms",
5069 "numeric",
5070 "measure",
5071 ),
5072 ("IA_LAST_RUN_DWELL_TIME", "last_run_dwell_time_ms", "numeric", "measure"),
5073 ("IA_FIXATION_COUNT", "n_fixations", "numeric", "measure"),
5074 ("IA_SKIP", "skip_flag", "boolean", "measure"),
5075 (
5076 "IA_REGRESSION_IN_COUNT",
5077 "number_of_regressions_in",
5078 "numeric",
5079 "measure",
5080 ),
5081 ("IA_REGRESSION_IN_COUNT", "regression_in_count", "numeric", "measure"),
5082 ("IA_REGRESSION_OUT_COUNT", "regression_out_count", "numeric", "measure"),
5083 ("IA_REGRESSION_IN", "regression_in_flag", "boolean", "measure"),
5084 ("IA_REGRESSION_OUT", "regression_out_flag", "boolean", "measure"),
5085 (
5086 "IA_REGRESSION_PATH_DURATION",
5087 "regression_path_duration_ms",
5088 "numeric",
5089 "measure",
5090 ),
5091 ("TRIAL_DWELL_TIME", "trial_dwell_time_ms", "numeric", "measure"),
5092 ("TRIAL_FIXATION_COUNT", "trial_fixation_count", "numeric", "measure"),
5093 ("TRIAL_IA_COUNT", "trial_ia_count", "numeric", "measure"),
5094 # A property of the word, not of the reading — and Corpus Analysis' "Word
5095 # length" feature, so it stays pre-kept with the other linguistic fields.
5096 ("word_length", "word_length", "numeric", "linguistic"),
5097 (
5098 "word_length_no_punctuation",
5099 "word_length_no_punctuation",
5100 "numeric",
5101 "linguistic",
5102 ),
5103 ("gpt2_surprisal", "gpt2_surprisal", "numeric", "linguistic"),
5104 ("wordfreq_frequency", "wordfreq_frequency", "numeric", "linguistic"),
5105 ("subtlex_frequency", "subtlex_frequency", "numeric", "linguistic"),
5106 ("universal_pos", "universal_pos", "string", "linguistic"),
5107 ("ptb_pos", "ptb_pos", "string", "linguistic"),
5108 ("Reduced_POS", "reduced_pos", "string", "linguistic"),
5109 ("dependency_relation", "dependency_relation", "string", "linguistic"),
5110 ("morphological_features", "morphological_features", "string", "linguistic"),
5111 ("entity_type", "entity_type", "string", "linguistic"),
5112 ("head_word_index", "head_word_index", "numeric", "linguistic"),
5113 ("distance_to_head", "distance_to_head", "numeric", "linguistic"),
5114 ("left_dependents_count", "left_dependents_count", "numeric", "linguistic"),
5115 ("right_dependents_count", "right_dependents_count", "numeric", "linguistic"),
5116 ("sentence_id", "sentence_id", "passthrough", "linguistic"),
5117 ("SENTENCE_ID", "sentence_id", "passthrough", "linguistic"),
5118 ("right_to_left", "right_to_left", "boolean", "meta"),
5119 ("RIGHT_TO_LEFT", "right_to_left", "boolean", "meta"),
5120 (SOURCE_FILE_COLUMN, SOURCE_FILE_COLUMN, "passthrough", "meta"),
5121 ("TRIAL_INDEX", "TRIAL_INDEX", "passthrough", "meta"),
5122 ("trial_index", "trial_index", "passthrough", "meta"),
5123 ("article_batch", "article_batch", "passthrough", "meta"),
5124 ("article_id", "article_id", "passthrough", "meta"),
5125 ("difficulty_level", "difficulty_level", "passthrough", "meta"),
5126 ("article_title", "article_title", "passthrough", "meta"),
5127 ("question", "question", "passthrough", "meta"),
5128 ("question_preview", "question_preview", "boolean", "meta"),
5129 ("selected_answer", "selected_answer", "passthrough", "meta"),
5130 ("is_correct", "is_correct", "passthrough", "meta"),
5131 ("repeated_reading_trial", "repeated_reading_trial", "boolean", "meta"),
5132 ("critical_span_indices", "critical_span_indices", "passthrough", "meta"),
5133 ("distractor_span_indices", "distractor_span_indices", "passthrough", "meta"),
5134 ("aspan_ind_start", "aspan_ind_start", "passthrough", "meta"),
5135 ("aspan_ind_end", "aspan_ind_end", "passthrough", "meta"),
5136 ("dspan_ind_start", "dspan_ind_start", "passthrough", "meta"),
5137 ("dspan_ind_end", "dspan_ind_end", "passthrough", "meta"),
5138 ("is_in_aspan", "is_in_aspan", "boolean", "meta"),
5139 ("is_in_dspan", "is_in_dspan", "boolean", "meta"),
5140 # MultiplEYE side-data (also see FIX_OPTIONAL_FIELDS): the comprehension
5141 # questions JSON + the per-trial stimulus-image path + the genre facet, kept
5142 # so the panels / image layer can read them off the word frame too.
5143 ("comprehension_questions", "comprehension_questions", "passthrough", "meta"),
5144 ("image_path", "image_path", "passthrough", "meta"),
5145 ("image_x", "image_x", "numeric", "meta"),
5146 ("image_y", "image_y", "numeric", "meta"),
5147 # Stimulus typeface (size in monitor px + CSS family) the images were rendered
5148 # with — the app snaps its font controls to these so the reading text matches.
5149 ("stimulus_font_px", "stimulus_font_px", "numeric", "meta"),
5150 ("stimulus_font_family", "stimulus_font_family", "passthrough", "meta"),
5151 ("genre", "genre", "string", "meta"),
5152 # DATA-24 multipart screens: what a screen *is* (`reading` / `question` on
5153 # MultiplEYE — a trial mixes both, and their fixations have different
5154 # provenance), and which block of a question screen a word belongs to
5155 # (`stem` / `target` / `distractor_a`… — a per-word facet).
5156 ("screen_kind", "screen_kind", "passthrough", "meta"),
5157 ("aoi_block", "aoi_block", "passthrough", "meta"),
5158 # DATA-27: which tier the EyeGenBench word boxes came from --
5159 # "real" | "reconstructed" | "synthesized". Carried so the UI can badge a
5160 # reconstructed layout rather than pass it off as the original screen.
5161 ("geometry_source", "geometry_source", "passthrough", "meta"),
5162]
5164FIX_OPTIONAL_FIELDS = [
5165 (SOURCE_FILE_COLUMN, SOURCE_FILE_COLUMN, "passthrough", "meta"),
5166 ("TRIAL_INDEX", "TRIAL_INDEX", "passthrough", "meta"),
5167 ("trial_index", "trial_index", "passthrough", "meta"),
5168 ("article_batch", "article_batch", "passthrough", "meta"),
5169 ("article_id", "article_id", "passthrough", "meta"),
5170 ("difficulty_level", "difficulty_level", "passthrough", "meta"),
5171 ("article_title", "article_title", "passthrough", "meta"),
5172 ("question", "question", "passthrough", "meta"),
5173 ("selected_answer", "selected_answer", "passthrough", "meta"),
5174 ("is_correct", "is_correct", "passthrough", "meta"),
5175 ("repeated_reading_trial", "repeated_reading_trial", "boolean", "meta"),
5176 ("question_preview", "question_preview", "boolean", "meta"),
5177 # Per-fixation extras — auto-detected + kept (renamed to canonical so the
5178 # colour-by / per-fixation filters still find them), but not schema mapping
5179 # fields. saccade_amplitude is also recomputed from X/Y by measures when
5180 # absent, so it's never lost.
5181 ("pass_index", "pass_index", "numeric", "fixation"),
5182 ("reread", "pass_index", "numeric", "fixation"),
5183 ("saccade_type", "saccade_type", "string", "fixation"),
5184 ("NEXT_SAC_DIRECTION", "saccade_type", "string", "fixation"),
5185 # BUG-25: `saccade_amplitude` is *pixels* — either the source column of that
5186 # name, or (when absent) the euclidean distance measures.py computes from
5187 # X/Y. EyeLink's two amplitude columns are **degrees of visual angle**, and
5188 # they are two *different* saccades — the one leaving this fixation and the
5189 # one that arrived at it — so each keeps its own canonical name with the unit
5190 # in it. They are deliberately NOT aliased onto `saccade_amplitude`: that
5191 # made one column mean px or deg depending on which columns the export
5192 # happened to carry (~78x apart on the bundled demo), under a hard-coded
5193 # "px" label. Converting instead would need `pixels_per_degree`, which is
5194 # ASSUMED on every built-in corpus.
5195 ("saccade_amplitude", "saccade_amplitude", "numeric", "fixation"),
5196 ("NEXT_SAC_AMPLITUDE", "next_saccade_amplitude_deg", "numeric", "fixation"),
5197 ("PREVIOUS_SAC_AMPLITUDE", "prev_saccade_amplitude_deg", "numeric", "fixation"),
5198 ("eye", "eye", "string", "fixation"),
5199 ("EYE_USED", "eye", "string", "fixation"),
5200 ("EYE_TRACKED", "eye", "string", "fixation"),
5201 ("is_blink", "is_blink", "boolean", "fixation"),
5202 ("blink", "is_blink", "boolean", "fixation"),
5203 ("blink_flag", "is_blink", "boolean", "fixation"),
5204 ("BLINK", "is_blink", "boolean", "fixation"),
5205 # MultiplEYE trial-level facets + side-data → Trial Info chips / filter
5206 # facets / the comprehension panel / the stimulus-image layer. All are
5207 # MultiplEYE-specific source names (carried only when the loader emits them),
5208 # so they're inert for other corpora.
5209 ("genre", "genre", "string", "meta"),
5210 ("session", "session", "string", "meta"),
5211 ("participant", "participant", "string", "meta"),
5212 ("is_practice", "is_practice", "boolean", "meta"),
5213 ("trial_num", "trial_num", "numeric", "meta"),
5214 ("comprehension_questions", "comprehension_questions", "passthrough", "meta"),
5215 ("image_path", "image_path", "passthrough", "meta"),
5216 ("image_x", "image_x", "numeric", "meta"),
5217 ("image_y", "image_y", "numeric", "meta"),
5218 ("stimulus_font_px", "stimulus_font_px", "numeric", "meta"),
5219 ("stimulus_font_family", "stimulus_font_family", "passthrough", "meta"),
5220 # Reader metadata merged from participant_data.csv (namespaced pp_*).
5221 ("pp_age", "pp_age", "numeric", "meta"),
5222 ("pp_gender", "pp_gender", "string", "meta"),
5223 ("pp_native_language", "pp_native_language", "string", "meta"),
5224 ("pp_years_education", "pp_years_education", "numeric", "meta"),
5225 ("pp_education_level", "pp_education_level", "string", "meta"),
5226 # DATA-24: which kind of screen a fixation happened on (see the word table).
5227 ("screen_kind", "screen_kind", "passthrough", "meta"),
5228 # DATA-27: which tier the EyeGenBench word boxes came from --
5229 # "real" | "reconstructed" | "synthesized". Carried so the UI can badge a
5230 # reconstructed layout rather than pass it off as the original screen.
5231 ("geometry_source", "geometry_source", "passthrough", "meta"),
5232 # DATA-31: whether EyeGenBench retained the recorded vertical coordinate or
5233 # placed this fixation at its word box's centre.
5234 ("fixation_y_source", "fixation_y_source", "passthrough", "meta"),
5235 # DATA-27: EyeGenBench's own composite trial id, kept for traceability back to the
5236 # benchmark. Deliberately NOT named `unique_trial_id` — normalize_fixations used to
5237 # key trial_id on any column with that literal name (BUG-58), and the normalized
5238 # frame's own `unique_trial_id` is the mapped trial id, so these would not survive.
5239 ("eyegenbench_trial_id", "eyegenbench_trial_id", "passthrough", "meta"),
5240]
5243def _schema_source_columns(schema: dict) -> set:
5244 """Set of raw source column names a normalization schema references."""
5245 cols: set = set()
5246 for value in schema.values():
5247 if not value:
5248 continue
5249 if isinstance(value, list):
5250 cols.update(value)
5251 else:
5252 cols.add(value)
5253 return cols
5256def dropped_columns(
5257 raw: pd.DataFrame,
5258 *,
5259 keep: set | None = None,
5260 schema: dict | None = None,
5261) -> list:
5262 """Original source columns discarded during normalization (sorted).
5264 Pass ``keep`` (the set handed to ``normalize_words``/``normalize_fixations``,
5265 i.e. a ``compute_keep_columns`` result) for the union-keep tables, or
5266 ``schema`` for raw gaze (``normalize_raw_gaze`` keeps the schema-referenced
5267 columns plus any ``unique_trial_id`` it consults directly). With neither,
5268 returns ``[]``."""
5269 if keep is None and schema is not None:
5270 keep = _schema_source_columns(schema) | {"unique_trial_id"}
5271 if keep is None:
5272 return []
5273 return sorted(c for c in raw.columns if c not in keep)
5276# EyeLink writes a missing value as the string ``'.'`` and booleans as ``'0'`` /
5277# ``'1'``, so a whole flag column arrives as *strings* (BUG-7). A plain
5278# ``astype(bool)`` then reads every non-empty string as True — including ``'0'``
5279# and ``'.'`` — and `regression_in_flag` came out True for every row in the
5280# bundled demo. Anything a reader would write for "false" or "missing" has to be
5281# recognised before the cast.
5282_FALSEY_FLAG_STRINGS = {"", ".", "0", "0.0", "false", "f", "no", "n", "na", "nan", "-"}
5285def coerce_flag(col: pd.Series) -> pd.Series:
5286 """Coerce a flag-like column to real booleans (BUG-7).
5288 Numbers go by ``!= 0``; strings are matched against the sentinels above
5289 (case-insensitively) rather than by truthiness. NaN / missing is ``False``,
5290 which is what every downstream flag consumer already assumed.
5291 """
5292 if pd.api.types.is_bool_dtype(col):
5293 return col.fillna(False).astype(bool)
5294 numeric = pd.to_numeric(col, errors="coerce")
5295 if numeric.notna().any():
5296 # A numeric-looking column ('0'/'1', 0/1, 0.0/1.0): non-zero is True.
5297 # Values that didn't parse fall through to the string test below, so a
5298 # mixed '0'/'1'/'.' column doesn't lose its '.' rows to True.
5299 parsed = numeric.notna()
5300 # Build the boolean array positionally: assigning a bool ndarray into a
5301 # bool Series by mask is deprecated in pandas as a dtype-incompatible set.
5302 values = (numeric.to_numpy() != 0) & parsed.to_numpy()
5303 unparsed = (~parsed & col.notna()).to_numpy()
5304 if unparsed.any():
5305 values[unparsed] = ~(
5306 col[unparsed]
5307 .astype(str)
5308 .str.strip()
5309 .str.lower()
5310 .isin(_FALSEY_FLAG_STRINGS)
5311 ).to_numpy()
5312 result = pd.Series(values, index=col.index, dtype=bool)
5313 return result
5314 return (
5315 ~col.fillna("").astype(str).str.strip().str.lower().isin(_FALSEY_FLAG_STRINGS)
5316 ).astype(bool)
5319#: What a supplied reading-measure flag writes for "not recorded" — EyeLink's
5320#: `.` for a word with no first pass, an empty cell, a spelled-out NA.
5321_MISSING_FLAG_STRINGS = {"", ".", "na", "nan", "n/a", "-", "none", "null", "<na>"}
5324_TRUE_FLAG_SPELLINGS = {"true", "t", "yes", "y", "1", "1.0"}
5325_FALSE_FLAG_SPELLINGS = {"false", "f", "no", "n", "0", "0.0"}
5326_FLAG_SPELLINGS = {
5327 **dict.fromkeys(_TRUE_FLAG_SPELLINGS, True),
5328 **dict.fromkeys(_FALSE_FLAG_SPELLINGS, False),
5329}
5332def coerce_bool_or_na(col: pd.Series) -> pd.Series:
5333 """A user-supplied true/false column as a nullable boolean.
5335 Real booleans stay as they are, numbers go by ``!= 0``, and the strings
5336 ``true/false``, ``t/f``, ``yes/no``, ``y/n``, ``1/0`` (any case, trimmed)
5337 are read by their meaning — never by truthiness, under which the string
5338 ``"False"`` is true. Missing cells and any other spelling are ``<NA>``, so
5339 a caller decides what an unknown means (round 11).
5340 """
5341 if pd.api.types.is_bool_dtype(col):
5342 return col.astype("boolean")
5343 if pd.api.types.is_numeric_dtype(col):
5344 return (col != 0).astype("boolean").mask(col.isna())
5345 out = pd.Series(pd.NA, index=col.index, dtype="boolean")
5346 is_str = col.map(type).eq(str).to_numpy()
5347 is_bool = col.map(type).eq(bool).to_numpy()
5348 if is_bool.any():
5349 out[is_bool] = col[is_bool].astype(bool).to_numpy()
5350 numeric = pd.to_numeric(col.where(~is_str & ~is_bool), errors="coerce")
5351 is_number = numeric.notna().to_numpy()
5352 out[is_number] = (numeric[is_number] != 0).to_numpy()
5353 if is_str.any():
5354 spelled = col[is_str].str.strip().str.lower()
5355 out[is_str] = spelled.map(_FLAG_SPELLINGS).astype("boolean").to_numpy()
5356 return out
5359def coerce_measure_flag(col: pd.Series) -> pd.Series:
5360 """A supplied reading-measure flag (skip, regression in/out) as a nullable
5361 boolean: true, false, or missing.
5363 :func:`coerce_flag` reads a missing cell as ``False``, which is right for an
5364 operational flag (blink, excluded) and wrong for a measure: EyeLink writes
5365 ``.`` in ``IA_REGRESSION_IN`` for a word with no first pass, where the
5366 measure is undefined, not "no regression". As ``False`` those rows lowered
5367 every rate and counted as readers behind it."""
5368 flags = coerce_flag(col).astype("boolean")
5369 if pd.api.types.is_bool_dtype(col) and not col.isna().any():
5370 return flags
5371 missing = col.isna() | col.astype(str).str.strip().str.lower().isin(
5372 _MISSING_FLAG_STRINGS
5373 )
5374 flags[missing.to_numpy()] = pd.NA
5375 return flags
5378def _apply_optional_fields(
5379 df: pd.DataFrame, source: pd.DataFrame, registry: list, keep: set | None
5380) -> set:
5381 """Carry registry-listed optional source columns into ``df`` (renamed +
5382 dtype-coerced). ``keep`` is ``None`` (carry every detected field — the
5383 backward-compatible default) or a set of *source* column names to limit to.
5384 Returns the set of source columns actually emitted."""
5385 emitted: set = set()
5386 for src, dest, kind, category in registry:
5387 if src not in source.columns:
5388 continue
5389 if keep is not None and src not in keep:
5390 continue
5391 emitted.add(src)
5392 col = source[src]
5393 if kind == "numeric":
5394 df[dest] = _to_number(col)
5395 elif kind == "string":
5396 df[dest] = col.astype(str)
5397 elif kind == "boolean" and category == "measure":
5398 # A reading measure keeps "not recorded" apart from "false".
5399 df[dest] = coerce_measure_flag(col)
5400 elif kind == "boolean":
5401 df[dest] = coerce_flag(col)
5402 else:
5403 df[dest] = col
5404 return emitted
5407def _carry_extra_columns(
5408 df: pd.DataFrame, source: pd.DataFrame, keep: set | None, skip: set
5409) -> None:
5410 """Carry user-chosen extra ``keep`` source columns through verbatim, skipping
5411 those already emitted (canonical / registry) or in ``skip``."""
5412 if not keep:
5413 return
5414 for col in keep:
5415 if col in source.columns and col not in skip and col not in df.columns:
5416 df[col] = source[col].to_numpy()
5419def categorize_columns(raw: pd.DataFrame, schema: dict, registry: list) -> dict:
5420 """Split a raw frame's columns into {mapped, detected_optional, unclaimed}.
5422 ``mapped`` = source columns the schema references; ``detected_optional`` =
5423 registry entries present in the frame (each ``{source, dest, category}``);
5424 ``unclaimed`` = everything else (offered as filter fields / extra keeps).
5426 AN-32: a registry column mapped as a *reading measure* is not also a
5427 detected extra — `IA_DWELL_TIME` mapped as TFD was offered, pre-kept, as
5428 `total_fixation_duration_ms` — and a source the registry lists twice (a
5429 compatibility alias) is detected once, under its first entry. Only the
5430 measure mapping claims a column here: a registry column the Trial ID is
5431 composed from (`repeated_reading_trial`) is still a detected trial
5432 condition, which is what offers it as a trial filter."""
5433 mapped = {c for c in _schema_source_columns(schema) if c in raw.columns}
5434 as_measure = {schema.get(key) for key in READING_MEASURE_KEYS if schema.get(key)}
5435 detected: list = []
5436 seen: set = set()
5437 for src, dest, _kind, category in registry:
5438 if src in raw.columns and src not in as_measure and src not in seen:
5439 seen.add(src)
5440 detected.append({"source": src, "dest": dest, "category": category})
5441 detected_sources = {d["source"] for d in detected}
5442 unclaimed = [
5443 c for c in raw.columns if c not in mapped and c not in detected_sources
5444 ]
5445 return {"mapped": mapped, "detected_optional": detected, "unclaimed": unclaimed}
5448def compute_keep_columns(
5449 schema: dict,
5450 *,
5451 optional_sources: Iterable[str] | None = None,
5452 filter_fields: Iterable[str] | None = None,
5453 keep_columns: Iterable[str] | None = None,
5454) -> set:
5455 """Source columns to retain before normalization (everything else is dropped
5456 for speed). Union of: schema-mapped sources, always-kept structural columns,
5457 chosen optional fields, chosen filter fields, and extra keep columns."""
5458 keep = set(_schema_source_columns(schema))
5459 # Structural columns consulted directly by normalize_* (not via schema).
5460 for col in (
5461 SOURCE_FILE_COLUMN,
5462 "unique_trial_id",
5463 "unique_paragraph_id",
5464 "TRIAL_INDEX",
5465 "trial_index",
5466 ):
5467 keep.add(col)
5468 for group in (optional_sources, filter_fields, keep_columns):
5469 if group:
5470 keep.update(group)
5471 return keep
5474def _copy_screen_fields(
5475 df: pd.DataFrame, source: pd.DataFrame, schema: dict
5476) -> pd.DataFrame:
5477 """Copy mapped part identity/metadata and normalize it in one place."""
5478 fields = (
5479 ("screen_id", SCREEN_ID, False),
5480 ("screen_index", SCREEN_INDEX, True),
5481 ("screen_timestamp", SCREEN_TIMESTAMP, True),
5482 ("screen_fixation_id", SCREEN_FIXATION_ID, False),
5483 ("canvas_width", CANVAS_WIDTH, True),
5484 ("canvas_height", CANVAS_HEIGHT, True),
5485 )
5486 for schema_key, destination, numeric in fields:
5487 column = schema.get(schema_key)
5488 if not column:
5489 continue
5490 values = source[column]
5491 df[destination] = _to_number(values) if numeric else values
5492 # BUG-79: UX-88 took `screen_index` out of the mapping on the premise that
5493 # the public corpora stamp it onto their frames — but this function rebuilds
5494 # the frame from the mapping, so the stamp was dropped and screen order
5495 # re-derived from row order: AOI-file order on the words, each reader's
5496 # onset order on the fixations. MultiplEYE's per-reader question order then
5497 # conflicted and the 🗂️ Data page crashed. A canonical column rides through
5498 # — but only onto a frame the mapping made multipart (DATA-59). With no
5499 # screen field mapped, a raw `screen_index` column is just a column: riding
5500 # it through derived a `screen_id` from it, so clearing the screen fields
5501 # in the mapping still made the AOI table multipart while the fixations
5502 # were not, and the pair was refused.
5503 if (
5504 SCREEN_ID in df.columns
5505 and SCREEN_INDEX not in df.columns
5506 and SCREEN_INDEX in source.columns
5507 ):
5508 df[SCREEN_INDEX] = _to_number(source[SCREEN_INDEX])
5509 return normalize_screen_identity(df)
5512def _text_id_mapped_flag(
5513 source: pd.DataFrame, schema: dict, *, renormalizing: bool
5514) -> bool | np.ndarray | None:
5515 """What ``normalize_*`` write to ``_text_id_mapped`` (``None``: nothing).
5517 A schema Text ID — auto-detected or picked — or a literal
5518 ``unique_paragraph_id`` is mapped. On ✏️ Edit dataset the proposal maps
5519 Text ID to the stored ``text_id`` column itself, which says nothing new: the
5520 stored frame's own flag is kept, and a frame stored without one gets none
5521 (so its values decide, as before)."""
5522 text = schema.get("text_id")
5523 if renormalizing and text and trial_mapping_columns(text) == ["text_id"]:
5524 if TEXT_ID_MAPPED in source.columns:
5525 return source[TEXT_ID_MAPPED].fillna(False).astype(bool).to_numpy()
5526 return None
5527 return bool(text) or "unique_paragraph_id" in source.columns
5530def _drop_reserved_columns(
5531 raw: pd.DataFrame, schema: dict, *, table: str
5532) -> pd.DataFrame:
5533 """Drop incoming columns named like :data:`INTERNAL_COLUMNS`, quietly.
5535 The pipeline reads those names as its own bookkeeping — a join route, the
5536 stimulus flag, the Edit-dataset collapse key — so an incoming column that
5537 carries one must not reach any of them (DATA-49); normalization rebuilds
5538 them. The names are the pipeline's own, so the usual source is a table
5539 Scanpath Studio wrote being loaded again, a normal round-trip: logged at
5540 INFO, not warned about. One the mapping names is the user's data and is
5541 left alone."""
5542 referenced = _schema_source_columns(schema)
5543 clash = sorted(
5544 c for c in INTERNAL_COLUMNS if c in raw.columns and c not in referenced
5545 )
5546 if not clash:
5547 return raw
5548 _LOGGER.info(
5549 "%s: dropped %s, Scanpath Studio's own bookkeeping column(s); they are "
5550 "rebuilt on load.",
5551 table,
5552 ", ".join(repr(c) for c in clash),
5553 )
5554 return raw.drop(columns=clash)
5557def normalize_words(
5558 words: pd.DataFrame,
5559 schema: dict[str, str],
5560 *,
5561 keep_columns: set | None = None,
5562 _renormalizing: bool = False,
5563) -> pd.DataFrame:
5564 if not _renormalizing:
5565 words = _drop_reserved_columns(words, schema, table="Words table")
5566 _warn_normalization_issues(words, schema, table="Words table")
5567 words = _drop_rows_missing_identity(words, schema)
5568 # The explicit index makes scalar assignments (e.g. the stimulus-level
5569 # participant placeholder) fill every row even when assigned first.
5570 df = pd.DataFrame(index=words.index)
5571 if schema.get("participant"):
5572 # str or list (a composite participant id, joined like the trial id).
5573 df["participant_id"] = trial_id_series(words, schema["participant"])
5574 else:
5575 # Stimulus-level word/AoI table (one row per word per text, shared by
5576 # all participants) — broadcast_stimulus_words() expands it across the
5577 # participants found in the fixations.
5578 df["participant_id"] = STIMULUS_PARTICIPANT
5579 df[STIMULUS_WORDS_FLAG] = True
5580 trial_cols = trial_mapping_columns(schema["trial"])
5581 if len(trial_cols) > 1:
5582 # User-composed unique trial ID: authoritative, so it wins over a raw
5583 # `unique_trial_id` column and needs no repeated-reading suffixing.
5584 df["trial_id"] = trial_id_series(words, trial_cols)
5585 df["unique_trial_id"] = df["trial_id"]
5586 unsuffixed = df["trial_id"]
5587 else:
5588 # The mapped column, always (BUG-58). A literal `unique_trial_id`
5589 # column used to win over whatever the mapping named, so a Trial ID
5590 # picked by hand was silently replaced on any table that carried one —
5591 # and a pair where only one side did joined on nothing. Auto-detection
5592 # proposes `unique_trial_id` first, so it is still used by default.
5593 trial_col = trial_cols[0]
5594 df["trial_id"] = stable_id(words[trial_col])
5595 # The id before any repeat suffix names the text that was read.
5596 unsuffixed = df["trial_id"]
5597 if schema.get("participant"):
5598 df = _disambiguate_repeated_readings(df, words, trial_col)
5599 if "unique_trial_id" in words.columns:
5600 # The mapped id *is* the unique trial id (BUG-58) — never the raw
5601 # column's own values, which the trial picker would otherwise key
5602 # on (`utils.build_combo_options` prefers `unique_trial_id`).
5603 df["unique_trial_id"] = df["trial_id"]
5604 if "unique_paragraph_id" in words.columns:
5605 df["unique_text_id"] = stable_id(words["unique_paragraph_id"])
5606 df["text_id"] = df["unique_text_id"]
5607 elif schema.get("text_id"):
5608 # str or list (a composite text id, joined like the trial id).
5609 df["text_id"] = trial_id_series(words, schema["text_id"])
5610 else:
5611 # DATA-49: a repeated reading's text is the id it was suffixed from —
5612 # the text a stimulus-level AOI table knows it by.
5613 df["text_id"] = unsuffixed
5614 mapped = _text_id_mapped_flag(words, schema, renormalizing=_renormalizing)
5615 if mapped is not None:
5616 df[TEXT_ID_MAPPED] = mapped
5617 df = _copy_screen_fields(df, words, schema)
5618 df["word_id"] = _to_number(words[schema["word_id"]])
5619 if schema.get("text"):
5620 # BUG-53: a missing cell is an empty word, never NaN — pandas 3's
5621 # `astype(str)` keeps NaN as NaN, and every " ".join over a trial's text
5622 # downstream then raises on the float.
5623 text = words[schema["text"]]
5624 df["text"] = text.where(text.notna(), "").astype(str)
5625 else:
5626 df["text"] = df["word_id"].apply(lambda v: f"w{int(v)}" if pd.notna(v) else "")
5627 df["text"] = df["text"].str.replace(r"\s+", " ", regex=True).str.strip()
5628 if schema.get("line"):
5629 df["line_idx"] = _to_number(words[schema["line"]])
5630 else:
5631 df["line_idx"] = 1
5633 if all(schema.get(k) for k in ["x", "y", "width", "height"]):
5634 df["x"] = _to_number(words[schema["x"]])
5635 df["y"] = _to_number(words[schema["y"]])
5636 df["width"] = _to_number(words[schema["width"]])
5637 df["height"] = _to_number(words[schema["height"]])
5638 else:
5639 left = _to_number(words[schema["left"]])
5640 right = _to_number(words[schema["right"]])
5641 top = _to_number(words[schema["top"]])
5642 bottom = _to_number(words[schema["bottom"]])
5643 df["x"] = left
5644 df["y"] = top
5645 df["width"] = right - left
5646 df["height"] = bottom - top
5648 emitted = _apply_optional_fields(df, words, WORD_OPTIONAL_FIELDS, keep_columns)
5649 if keep_columns is not None:
5650 _carry_extra_columns(
5651 df, words, keep_columns, _schema_source_columns(schema) | emitted
5652 )
5653 # AN-32: after the passthrough and the extras, so the mapping has the last
5654 # word on every measure it names.
5655 _apply_reading_measures(df, words, schema)
5656 _blank_unfixated_measures(df)
5658 df = _preserve_composite_columns(df, words, schema["trial"])
5659 return df
5662def normalize_fixations(
5663 fixations: pd.DataFrame,
5664 schema: dict[str, str],
5665 *,
5666 keep_columns: set | None = None,
5667 _renormalizing: bool = False,
5668) -> pd.DataFrame:
5669 if not _renormalizing:
5670 fixations = _drop_reserved_columns(fixations, schema, table="Fixations")
5671 _warn_normalization_issues(fixations, schema, table="Fixations", fixations=True)
5672 fixations = _drop_rows_missing_identity(fixations, schema)
5673 # Explicit index so a constant participant placeholder fills every row.
5674 df = pd.DataFrame(index=fixations.index)
5675 if schema.get("participant"):
5676 # str or list (a composite participant id, joined like the trial id).
5677 df["participant_id"] = trial_id_series(fixations, schema["participant"])
5678 else:
5679 # No participant column → a single anonymous reader.
5680 df["participant_id"] = SYNTHETIC_PARTICIPANT
5681 trial_cols = trial_mapping_columns(schema["trial"])
5682 if len(trial_cols) > 1:
5683 # User-composed unique trial ID — see normalize_words.
5684 df["trial_id"] = trial_id_series(fixations, trial_cols)
5685 df["unique_trial_id"] = df["trial_id"]
5686 unsuffixed = df["trial_id"]
5687 else:
5688 trial_col = trial_cols[0] # the mapped column (BUG-58)
5689 df["trial_id"] = stable_id(fixations[trial_col])
5690 # The id before any repeat suffix names the text that was read.
5691 unsuffixed = df["trial_id"]
5692 if schema.get("participant"):
5693 df = _disambiguate_repeated_readings(
5694 df, fixations, trial_col, record_base=True
5695 )
5696 if "unique_trial_id" in fixations.columns:
5697 # The mapped id *is* the unique trial id (BUG-58) — never the raw
5698 # column's own values, which the trial picker would otherwise key
5699 # on (`utils.build_combo_options` prefers `unique_trial_id`).
5700 df["unique_trial_id"] = df["trial_id"]
5701 if "unique_paragraph_id" in fixations.columns:
5702 df["text_id"] = stable_id(fixations["unique_paragraph_id"])
5703 elif schema.get("text_id"):
5704 # str or list (a composite text id, joined like the trial id).
5705 df["text_id"] = trial_id_series(fixations, schema["text_id"])
5706 else:
5707 # DATA-49: the text a repeated reading is of, without its `_r2`.
5708 df["text_id"] = unsuffixed
5709 mapped = _text_id_mapped_flag(fixations, schema, renormalizing=_renormalizing)
5710 if mapped is not None:
5711 df[TEXT_ID_MAPPED] = mapped
5712 if "unique_paragraph_id" in fixations.columns:
5713 df["unique_text_id"] = stable_id(fixations["unique_paragraph_id"])
5714 df = _copy_screen_fields(df, fixations, schema)
5715 # X/Y may be unmapped for AOI-sequence datasets (no pixel coordinates) —
5716 # left NaN here and filled from word-box centers by harmonize_frames().
5717 for coord in ("x", "y"):
5718 if schema.get(coord):
5719 df[coord] = _to_number(fixations[schema[coord]])
5720 else:
5721 df[coord] = np.nan
5722 # An unreadable duration / onset still falls back to 0, but no longer
5723 # silently: `_warn_numeric_issues` below names the column (BUG-54).
5724 # DATA-40: a duration / onset in seconds (Gazepoint), microseconds (Tobii)
5725 # or nanoseconds (Pupil Labs Neon) is read in milliseconds.
5726 duration = schema["duration"]
5727 df["duration_ms"] = _as_ms(_to_number(fixations[duration]), duration).fillna(0)
5729 if schema.get("timestamp"):
5730 onset = schema["timestamp"]
5731 df["timestamp_ms"] = _as_ms(_to_number(fixations[onset]), onset).fillna(0)
5732 else:
5733 df["timestamp_ms"] = df.groupby(list(PARENT_KEY), sort=False).cumcount()
5735 if schema.get("fixation_id"):
5736 df["fixation_id"] = fixations[schema["fixation_id"]]
5737 else:
5738 df["fixation_id"] = df.groupby(list(PARENT_KEY), sort=False).cumcount().add(1)
5740 if SCREEN_ID in df.columns:
5741 part_keys = grouping_columns(df)
5742 if SCREEN_TIMESTAMP not in df.columns:
5743 df[SCREEN_TIMESTAMP] = df.groupby(part_keys, sort=False).cumcount()
5744 if SCREEN_FIXATION_ID not in df.columns:
5745 df[SCREEN_FIXATION_ID] = df.groupby(part_keys, sort=False).cumcount().add(1)
5747 if schema.get("word_id"):
5748 df["word_id"] = _to_number(fixations[schema["word_id"]])
5749 else:
5750 df["word_id"] = np.nan
5752 # pass_index / saccade_type / saccade_amplitude / eye are no longer schema
5753 # fields — they ride through _apply_optional_fields (FIX_OPTIONAL_FIELDS) when
5754 # the data carries them. saccade_amplitude is also recomputed from X/Y by
5755 # measures.enrich_fixations when absent.
5756 emitted = _apply_optional_fields(df, fixations, FIX_OPTIONAL_FIELDS, keep_columns)
5757 if keep_columns is not None:
5758 _carry_extra_columns(
5759 df, fixations, keep_columns, _schema_source_columns(schema) | emitted
5760 )
5762 df = _preserve_composite_columns(df, fixations, schema["trial"])
5763 synthesized = _synthesized_timestamps(fixations, schema, _renormalizing)
5764 if synthesized is None:
5765 df = df.drop(columns=[TIMESTAMP_SYNTHESIZED], errors="ignore")
5766 else:
5767 df[TIMESTAMP_SYNTHESIZED] = synthesized
5769 df["order_in_trial"] = (
5770 df.sort_values(["timestamp_ms", "duration_ms"])
5771 .groupby(list(PARENT_KEY), sort=False)
5772 .cumcount()
5773 + 1
5774 )
5775 if SCREEN_ID in df.columns:
5776 df["order_in_screen"] = (
5777 df.sort_values([SCREEN_INDEX, SCREEN_TIMESTAMP, "duration_ms"])
5778 .groupby(grouping_columns(df), sort=False)
5779 .cumcount()
5780 + 1
5781 )
5782 return df
5785# Canonical columns produced by normalize_words / normalize_fixations. Used to
5786# build typed empty frames when a dataset ships only one of the two reports,
5787# so every downstream consumer can keep selecting columns unconditionally.
5788WORDS_CANONICAL_COLUMNS: dict[str, str] = {
5789 "participant_id": "object",
5790 "trial_id": "object",
5791 "text_id": "object",
5792 "word_id": "float64",
5793 "text": "object",
5794 "line_idx": "float64",
5795 "x": "float64",
5796 "y": "float64",
5797 "width": "float64",
5798 "height": "float64",
5799}
5800FIX_CANONICAL_COLUMNS: dict[str, str] = {
5801 "participant_id": "object",
5802 "trial_id": "object",
5803 "text_id": "object",
5804 "x": "float64",
5805 "y": "float64",
5806 "duration_ms": "float64",
5807 "timestamp_ms": "float64",
5808 "fixation_id": "float64",
5809 "word_id": "float64",
5810 "order_in_trial": "int64",
5811}
5814def empty_words_frame() -> pd.DataFrame:
5815 """An empty words frame with the canonical post-normalization columns."""
5816 return pd.DataFrame(
5817 {col: pd.Series(dtype=dt) for col, dt in WORDS_CANONICAL_COLUMNS.items()}
5818 )
5821def empty_fixations_frame() -> pd.DataFrame:
5822 """An empty fixations frame with the canonical post-normalization columns."""
5823 return pd.DataFrame(
5824 {col: pd.Series(dtype=dt) for col, dt in FIX_CANONICAL_COLUMNS.items()}
5825 )
5828# Identity columns recomputed (not carried) by a remap — see remap_normalized_frame.
5829_REMAP_DERIVED_IDS = ("unique_trial_id", "unique_text_id", "unique_paragraph_id")
5832def repeat_bases(fixations: pd.DataFrame | None) -> dict:
5833 """Each repeated reading's ``(reader, trial id)`` → the trial id it was
5834 recorded under.
5836 From ``_base_trial_id`` when the fixations carry it. A stored frame from
5837 before DATA-49 does not, so its suffixes are read back instead: a trial
5838 ``X_rN`` of a reader who also has a trial ``X`` is the repeat
5839 ``_disambiguate_repeated_readings`` made of it. Only the pre-provenance
5840 Edit-dataset path uses the second form, to fold a stored repeat's copy of
5841 the boxes back into the AOI trial it copied."""
5842 if fixations is None or fixations.empty or "trial_id" not in fixations:
5843 return {}
5844 if "participant_id" not in fixations.columns:
5845 return {}
5846 if BASE_TRIAL_ID in fixations.columns:
5847 pairs = fixations[["participant_id", "trial_id", BASE_TRIAL_ID]]
5848 pairs = pairs.dropna().drop_duplicates().astype(str)
5849 pairs = pairs[pairs["trial_id"] != pairs[BASE_TRIAL_ID]]
5850 return dict(
5851 zip(
5852 zip(pairs["participant_id"], pairs["trial_id"]),
5853 pairs[BASE_TRIAL_ID],
5854 )
5855 )
5856 keys = fixations[["participant_id", "trial_id"]].dropna().drop_duplicates()
5857 keys = keys.astype(str)
5858 base = keys["trial_id"].str.extract(_REPEAT_SUFFIX, expand=False)
5859 candidates = keys.assign(base=base).dropna(subset=["base"])
5860 have = pd.MultiIndex.from_frame(keys[["participant_id", "trial_id"]])
5861 known = pd.MultiIndex.from_arrays(
5862 [candidates["participant_id"], candidates["base"]]
5863 ).isin(have)
5864 chosen = candidates[known]
5865 return dict(zip(zip(chosen["participant_id"], chosen["trial_id"]), chosen["base"]))
5868#: A repeated reading's suffix exactly as `_disambiguate_repeated_readings`
5869#: writes it: `_r2`, `_r3` …, never `_r0`/`_r1` or a zero-padded number.
5870_REPEAT_SUFFIX = r"^(.+)_r(?:[2-9]|[1-9]\d+)$"
5873def _map_repeats(frame: pd.DataFrame, repeat_of: dict) -> pd.Series:
5874 """``frame``'s trial ids with each repeat replaced by its base, looked up
5875 per ``(reader, trial)`` in a :func:`repeat_bases` mapping."""
5876 ids = frame["trial_id"].astype(str)
5877 if not repeat_of or "participant_id" not in frame.columns:
5878 return ids
5879 table = pd.Series(
5880 list(repeat_of.values()), index=pd.MultiIndex.from_tuples(list(repeat_of))
5881 )
5882 wanted = pd.MultiIndex.from_arrays([frame["participant_id"].astype(str), ids])
5883 found = table.reindex(wanted).to_numpy()
5884 return pd.Series(np.where(pd.isna(found), ids.to_numpy(), found), index=frame.index)
5887def _synthesized_timestamps(
5888 fixations: pd.DataFrame, schema: dict, renormalizing: bool
5889) -> pd.Series | None:
5890 """The :data:`TIMESTAMP_SYNTHESIZED` flags for a normalized fixations frame.
5892 Every row when no onset is mapped. A remap that keeps the stored
5893 ``timestamp_ms`` as the onset keeps the stored flags — those numbers are
5894 still the ones normalization made up. ``None`` when the timestamps are
5895 the data's own."""
5896 onset = schema.get("timestamp")
5897 if not onset:
5898 return pd.Series(True, index=fixations.index)
5899 if (
5900 renormalizing
5901 and onset == "timestamp_ms"
5902 and TIMESTAMP_SYNTHESIZED in fixations.columns
5903 ):
5904 return fixations[TIMESTAMP_SYNTHESIZED].fillna(False).astype(bool)
5905 return None
5908def remap_normalized_frame(
5909 frame: pd.DataFrame,
5910 schema: dict[str, str | None],
5911 *,
5912 kind: str,
5913 repeat_of: dict | None = None,
5914) -> pd.DataFrame:
5915 """Re-derive an already-normalized frame under a new column mapping.
5917 Stored datasets keep only their post-normalization frames — canonical column
5918 names (``duration_ms``, ``x``, ``word_id``, …) plus any kept extras; the
5919 original upload columns are gone. To change the mapping without re-uploading,
5920 re-run the matching ``normalize_*`` over the *normalized* frame, treating its
5921 current columns as the source universe. ``schema`` therefore references
5922 canonical/extra column names (e.g. ``{"duration": "duration_ms",
5923 "trial": "trial_id", ...}``).
5925 The precomputed identity columns (``unique_trial_id`` etc.) are dropped first
5926 so the new Trial/Text mapping is authoritative: otherwise ``normalize_*``
5927 would keep deriving ``trial_id`` from the existing ``unique_trial_id`` and
5928 silently ignore a changed Trial ID pick. A derived-id column the new schema
5929 *references* is NOT dropped, though — a composite trial id can be built from
5930 ``unique_paragraph_id``, and dropping a chosen component would make
5931 ``trial_id_series`` raise ``KeyError``. Every surviving column is kept
5932 (``keep_columns`` = all current columns) so the remap only reassigns roles
5933 and never drops data that already survived the first normalization. After
5934 normalization ``unique_trial_id`` / ``unique_text_id`` are restored
5935 (= ``trial_id`` / ``text_id``) when the single-column path didn't set them,
5936 so the frame's identity columns stay consistent with the composite path and
5937 downstream readers of ``unique_text_id`` keep working.
5939 DATA-39: a **words** frame remapped with no Participant is a stimulus-level
5940 AOI table, exactly as it was at import — ``normalize_words`` re-flags it for
5941 ``broadcast_stimulus_words``. But the stored frame was *already* broadcast
5942 (one copy of every trial's words per reader), so only the first reader's
5943 copy of each trial is kept here, and the caller must run
5944 ``harmonize_frames`` to broadcast it again. Skipping either step is the bug
5945 this fixes: without the collapse every reader gets every reader's boxes;
5946 without the harmonize every word is left on the ``""`` placeholder reader,
5947 so no trial finds its boxes and the scanpath plot loses its AOIs and its
5948 text. The collapse picks a *reader*, never a key: deduplicating on
5949 ``word_id`` would also merge rows that are not copies at all — character
5950 AOIs sharing a word id, or ids that do not parse as numbers and all fold to
5951 NaN."""
5952 referenced = _schema_source_columns(schema)
5953 derived = list(_REMAP_DERIVED_IDS)
5954 if trial_mapping_columns(schema.get("trial") or []) != ["trial_id"]:
5955 # A repeat's recorded id belongs to the trial mapping that suffixed it:
5956 # kept while the Trial ID stays the stored `trial_id` (the editor's own
5957 # proposal), re-derived by the disambiguation under any other pick.
5958 derived.append(BASE_TRIAL_ID)
5959 working = frame.drop(
5960 columns=[c for c in derived if c in frame.columns and c not in referenced]
5961 )
5962 if (
5963 kind == "fixations"
5964 and repeat_of
5965 and BASE_TRIAL_ID not in working.columns
5966 and BASE_TRIAL_ID not in derived
5967 ):
5968 # A fixations frame stored before `_base_trial_id` existed: give its
5969 # repeats the id they were recorded under, so the join still finds
5970 # their boxes after the save.
5971 bases = _map_repeats(working, repeat_of)
5972 if (bases != working["trial_id"].astype(str)).any():
5973 working = working.assign(**{BASE_TRIAL_ID: bases})
5974 if (
5975 kind == "words"
5976 and not schema.get("participant")
5977 and "participant_id" in working.columns
5978 and "trial_id" in working.columns
5979 and not working.empty
5980 ):
5981 if AOI_TRIAL_ID in working.columns:
5982 # Broadcast here (DATA-49): each copy names the AOI trial it came
5983 # from, whichever route it took — so one reading's copy of each AOI
5984 # trial, stamped back with that trial's own id, *is* the stimulus
5985 # table again. No guessing from content, no suffixed repeats left.
5986 copy_keys = [AOI_TRIAL_ID, *[SCREEN_ID] * (SCREEN_ID in working.columns)]
5987 reading = (
5988 working["participant_id"].astype(str)
5989 + "\x1f"
5990 + working["trial_id"].astype(str)
5991 )
5992 else:
5993 # Stored before the provenance column: joined by trial id, so the
5994 # readers of a stimulus share its trial id — except a repeat, whose
5995 # copy carries its `_r2` id. Fold it back into the trial it copied
5996 # (`repeat_of`, from the fixations), or it would become a phantom
5997 # AOI trial of its own and make its text ambiguous for good.
5998 # One reading's copy — its reader *and* its own (stored) trial id,
5999 # so a reader's repeat is not kept as a second copy.
6000 reading = (
6001 working["participant_id"].astype(str)
6002 + "\x1f"
6003 + working["trial_id"].astype(str)
6004 )
6005 if repeat_of:
6006 working = working.assign(trial_id=_map_repeats(working, repeat_of))
6007 copy_keys = [c for c in ("trial_id", SCREEN_ID) if c in working.columns]
6008 first = reading.groupby(
6009 [working[c] for c in copy_keys], dropna=False, sort=False
6010 ).transform("first")
6011 working = working[reading == first]
6012 if AOI_TRIAL_ID in working.columns:
6013 working = working.assign(trial_id=working[AOI_TRIAL_ID]).drop(
6014 columns=[AOI_TRIAL_ID]
6015 )
6016 keep = set(working.columns)
6017 if kind == "words":
6018 result = normalize_words(
6019 working, schema, keep_columns=keep, _renormalizing=True
6020 )
6021 elif kind == "fixations":
6022 result = normalize_fixations(
6023 working, schema, keep_columns=keep, _renormalizing=True
6024 )
6025 elif kind == "raw_gaze":
6026 result = normalize_raw_gaze(working, schema, keep_columns=keep)
6027 else:
6028 raise ValueError(f"unknown frame kind: {kind!r}")
6029 if "unique_trial_id" not in result.columns and "trial_id" in result.columns:
6030 result["unique_trial_id"] = result["trial_id"]
6031 if "unique_text_id" not in result.columns and "text_id" in result.columns:
6032 result["unique_text_id"] = result["text_id"]
6033 return result
6036def _union_column_values(
6037 words: pd.DataFrame, fixations: pd.DataFrame, column: str
6038) -> list:
6039 """Sorted union of a column's values across both frames (either may be
6040 empty — single-report datasets have words or fixations, not both)."""
6041 values: set = set()
6042 for df in (words, fixations):
6043 if df is not None and not df.empty and column in df.columns:
6044 values.update(df[column].unique())
6045 return sorted(values)
6048def filter_data(
6049 words: pd.DataFrame,
6050 fixations: pd.DataFrame,
6051 filters: dict,
6052) -> tuple[pd.DataFrame, pd.DataFrame]:
6053 # When the participant/trial selection covers the whole frame (the default —
6054 # any narrowing already happened upstream in filter_trials), skip the two
6055 # O(n) membership masks entirely; only the optional fixation-level filters
6056 # below apply. ``default_filters`` sets the cover-all flags.
6057 cover_all = bool(
6058 filters.get("_participants_cover_all") and filters.get("_trials_cover_all")
6059 )
6060 if cover_all:
6061 # participant/trial cover the whole frame and default_filters set the
6062 # pass/saccade/eye filters to their full value sets (no-ops), so nothing
6063 # can narrow the fixations — return the frames untouched (no full-frame
6064 # mask, no copy), the common large-upload case.
6065 return words, fixations
6067 # As in `filter_trials`: only a missing key means "every participant". An
6068 # explicit empty list is a narrowing that matches nobody.
6069 participants = filters.get("participants")
6070 if participants is None:
6071 participants = _union_column_values(words, fixations, "participant_id")
6072 trials = filters.get("trials") or _union_column_values(words, fixations, "trial_id")
6073 word_mask = words["participant_id"].isin(participants) & words["trial_id"].isin(
6074 trials
6075 )
6076 words_filtered = words[word_mask]
6077 fix_mask = fixations["participant_id"].isin(participants) & fixations[
6078 "trial_id"
6079 ].isin(trials)
6080 if "pass_index" in fixations.columns:
6081 pass_indices = filters.get("pass_indices")
6082 if pass_indices:
6083 fix_mask &= fixations["pass_index"].isin(pass_indices)
6084 if "saccade_type" in fixations.columns:
6085 saccade_types = filters.get("saccade_types")
6086 if saccade_types:
6087 fix_mask &= fixations["saccade_type"].isin(saccade_types)
6088 if "eye" in fixations.columns:
6089 eyes = filters.get("eyes")
6090 if eyes:
6091 fix_mask &= fixations["eye"].isin(eyes)
6092 fixations_filtered = fixations[fix_mask]
6093 return words_filtered, fixations_filtered
6096def filter_trials(
6097 words: pd.DataFrame,
6098 fixations: pd.DataFrame,
6099 participants: list | None = None,
6100 metadata: dict[str, set] | None = None,
6101 ranges: dict[str, tuple[float, float]] | None = None,
6102 drop_unknown: Iterable[str] | None = None,
6103) -> tuple[pd.DataFrame, pd.DataFrame]:
6104 """Narrow words + fixations by participant and by trial metadata.
6106 ``metadata`` maps a column name to the set of allowed values (membership).
6107 Only columns present on a frame are applied, so a condition like
6108 ``question_preview`` (Hunting/Gathering) narrows both words and fixations —
6109 the column is copied onto both during normalization. A falsy selection means
6110 "no constraint".
6112 ``ranges`` (UX-49) is the *continuous* counterpart: column → ``(lo, hi)``
6113 inclusive bounds for a numeric trial-level column, where enumerating the
6114 distinct values would be useless. **Rows with no value survive**: a range is
6115 a narrowing control, so a reader missing a comprehension score is not what
6116 the user asked to exclude — and pandas compares ``NaN`` as ``False``, so the
6117 obvious bare ``.between()`` would silently drop every one of them.
6118 ``drop_unknown`` names the ranged columns whose researcher unticked *Keep
6119 unknown values*: there, a row with no value is left out with the rest.
6120 """
6121 w, f = words, fixations
6122 dropping = set(drop_unknown or ())
6123 # `None` is "no constraint"; an **empty list is a constraint that nothing
6124 # satisfies** and must empty the pool. The two were conflated while every
6125 # producer could only emit None-or-non-empty, but DATA-20's metadata
6126 # narrowing intersects sets and can legitimately land on zero readers —
6127 # under the old falsy test that silently applied no filter at all, so an
6128 # impossible combination showed the *whole* corpus.
6129 if participants is not None:
6130 # participant_id is already string after normalization (as the metadata
6131 # filters below also assume), so skip a full-column .astype(str) recast.
6132 keep = set(map(str, participants))
6133 w = w[w["participant_id"].isin(keep)]
6134 f = f[f["participant_id"].isin(keep)]
6135 for col, allowed in (metadata or {}).items():
6136 if not allowed:
6137 continue
6138 allowed = set(allowed)
6139 if col in w.columns:
6140 w = w[w[col].isin(allowed)]
6141 if col in f.columns:
6142 f = f[f[col].isin(allowed)]
6143 for col, bounds in (ranges or {}).items():
6144 if not bounds:
6145 continue
6146 lo, hi = bounds
6147 for frame_name in ("w", "f"):
6148 frame = w if frame_name == "w" else f
6149 if col not in frame.columns:
6150 continue
6151 values = pd.to_numeric(frame[col], errors="coerce")
6152 mask = values.between(lo, hi)
6153 if col not in dropping:
6154 mask |= values.isna()
6155 if frame_name == "w":
6156 w = w[mask]
6157 else:
6158 f = f[mask]
6159 return w, f
6162def trial_keys(frame: pd.DataFrame) -> set:
6163 """The distinct ``(participant_id, trial_id)`` string keys a frame carries.
6165 Deduplicates before materializing the tuples, so it stays cheap on a
6166 sample-level table (raw gaze) where every trial spans thousands of rows.
6167 A missing/empty frame, or one without both id columns, yields an empty set.
6168 """
6169 if frame is None or frame.empty:
6170 return set()
6171 if "participant_id" not in frame.columns or "trial_id" not in frame.columns:
6172 return set()
6173 pairs = frame[["participant_id", "trial_id"]].drop_duplicates()
6174 return {
6175 (str(p), str(t)) for p, t in zip(pairs["participant_id"], pairs["trial_id"])
6176 }
6179def text_ids(*frames: pd.DataFrame | None) -> set[str]:
6180 """The distinct text ids across every frame that carries one (DATA-50).
6182 ``unique_text_id`` when any frame has it, else ``text_id`` — one id space,
6183 never a union of the two. Counting the words table alone said **0 texts**
6184 for a fixations-only dataset whose fixations name twelve.
6185 """
6186 present = [f for f in frames if f is not None and not f.empty]
6187 column = (
6188 "unique_text_id"
6189 if any("unique_text_id" in f.columns for f in present)
6190 else "text_id"
6191 )
6192 found: set[str] = set()
6193 for frame in present:
6194 if column in frame.columns:
6195 found.update(str(v) for v in frame[column].dropna().unique())
6196 return found
6199def filter_frame_to_keys(frame: pd.DataFrame, keys: set) -> pd.DataFrame:
6200 """Keep only rows whose ``(participant_id, trial_id)`` is in ``keys``.
6202 Single-frame counterpart of :func:`filter_to_keys`. BUG-12: the raw-gaze
6203 samples table has to be narrowed by the annotation filters (favorites /
6204 tags) exactly like the words and fixations frames, or a sample row for an
6205 unstarred trial survives "⭐ Favorites only". Vectorized via a MultiIndex
6206 membership test so it stays fast on large tables.
6207 """
6208 if frame is None or frame.empty:
6209 return frame
6210 if "participant_id" not in frame.columns or "trial_id" not in frame.columns:
6211 return frame
6212 idx = pd.MultiIndex.from_arrays(
6213 [frame["participant_id"].astype(str), frame["trial_id"].astype(str)]
6214 )
6215 return frame[idx.isin(keys)]
6218def filter_to_keys(
6219 words: pd.DataFrame,
6220 fixations: pd.DataFrame,
6221 keys: set,
6222) -> tuple[pd.DataFrame, pd.DataFrame]:
6223 """Keep only rows whose (participant_id, trial_id) is in ``keys``.
6225 ``keys`` is a set of ``(str, str)`` tuples. Used to apply annotation-based
6226 filtering (favorites / tags). Vectorized via a MultiIndex membership test so
6227 it stays fast on large fixation tables."""
6228 return filter_frame_to_keys(words, keys), filter_frame_to_keys(fixations, keys)
6231def raw_gaze_in_pool(
6232 raw_gaze: pd.DataFrame,
6233 words_all: pd.DataFrame,
6234 fixations_all: pd.DataFrame,
6235 words_pool: pd.DataFrame,
6236 fixations_pool: pd.DataFrame,
6237) -> pd.DataFrame:
6238 """The raw-gaze rows of the trials in the current pool (VIZ-45).
6240 A trial the words or fixations table knows is in the pool when it survived
6241 the filters there, so its samples follow it. A trial **only the raw gaze
6242 knows** — a raw-gaze-only dataset's every trial, or the samples-only trials
6243 of a dataset that has fixations for others — has nothing there to survive,
6244 and used to be dropped for it: the old narrowing kept only the participants
6245 and trials the other two tables listed. Those trials stay, narrowed only by
6246 what applies to the samples themselves (participant, annotations, the
6247 trial-metadata keys — applied to ``raw_gaze`` before this).
6249 ``words_all`` / ``fixations_all`` are the frames *before* the filters, which
6250 is what tells "filtered out" from "never there". Returns ``raw_gaze`` itself
6251 when nothing is dropped — always when the pools are those very frames (no
6252 filter set), with no scan at all — and otherwise the narrowed frame from a
6253 `frame_cache`, so a rerun under the same filters reuses it rather than
6254 re-masking every sample.
6255 """
6256 if raw_gaze is None or raw_gaze.empty:
6257 return pd.DataFrame()
6258 if (words_all is None or words_all.empty) and (
6259 fixations_all is None or fixations_all.empty
6260 ):
6261 return raw_gaze
6262 if words_pool is words_all and fixations_pool is fixations_all:
6263 # Unfiltered: every raw-gaze trial is either pooled or unknown to them.
6264 return raw_gaze
6265 key = tuple(
6266 frame_fingerprint(frame)
6267 for frame in (raw_gaze, words_all, fixations_all, words_pool, fixations_pool)
6268 )
6270 def _build() -> pd.DataFrame:
6271 known = _known_trial_keys(
6272 words_all,
6273 fixations_all,
6274 cache_key=(frame_fingerprint(words_all), frame_fingerprint(fixations_all)),
6275 )
6276 # Per raw-gaze table, not per filter change: `frame_cache` keeps one
6277 # entry per slot, so toggling a filter back rebuilds this — and the
6278 # samples' own keys are the one full pass worth not repeating.
6279 present = _raw_gaze_trial_keys(raw_gaze, cache_key=frame_fingerprint(raw_gaze))
6280 pooled = trial_keys(words_pool) | trial_keys(fixations_pool)
6281 keep = {k for k in present if k in pooled or k not in known}
6282 return raw_gaze if keep == present else filter_frame_to_keys(raw_gaze, keep)
6284 return frame_cache("raw_gaze_pool", key, _build)
6287@st.cache_data(show_spinner=False, max_entries=8)
6288def _raw_gaze_trial_keys(_raw_gaze: pd.DataFrame, cache_key) -> set:
6289 """`trial_keys` of one (narrowed) raw-gaze table, cached on its fingerprint."""
6290 progress.report()
6291 return trial_keys(_raw_gaze)
6294@st.cache_data(show_spinner=False, max_entries=8)
6295def _known_trial_keys(
6296 _words_all: pd.DataFrame, _fixations_all: pd.DataFrame, cache_key
6297) -> set:
6298 """The trials the unfiltered words and fixations know, once per dataset —
6299 not once per filter change (`raw_gaze_in_pool`)."""
6300 progress.report()
6301 return trial_keys(_words_all) | trial_keys(_fixations_all)
6304# ---------------------------------------------------------------------------
6305# VAL-7 — is a "trial" actually more than one reading?
6306#
6307# The inverse of BUG-23. A Trial ID mapping that does not fully identify a
6308# reading concatenates several readings into one `trial_id`, and nothing about
6309# the result looks wrong: the figure renders as an ordinary scanpath with a lot
6310# of regressions. Three independent signals catch it, and a fourth names the fix.
6311#
6312# THE KEY IS `(participant_id, trial_id, screen_id)` — the multipart scientific
6313# grouping key, not `(participant_id, trial_id)`. A legitimate two-screen trial
6314# restarts `word_id` per screen, so the words signal reports duplicates that are
6315# correct data when grouped by trial alone. The two fixation signals are immune
6316# either way, because multipart keeps `fixation_id` and `timestamp_ms`
6317# parent-global by design — which is exactly why they are the right cross-check
6318# rather than a redundant one.
6319# ---------------------------------------------------------------------------
6321#: Columns that should hold ONE value inside a reading, so a second value is
6322#: evidence that two readings were merged — and the column name *is* the remedy
6323#: ("add `article_id` to the Trial ID mapping"). Deliberately a fixed list of
6324#: identity/condition fields rather than every column: a per-word or
6325#: per-fixation column varies within a trial by design, and scanning them all
6326#: would bury the signal in noise.
6327_TRIAL_WITNESS_COLUMNS: tuple[str, ...] = (
6328 "TRIAL_INDEX",
6329 "trial_index",
6330 "trial_num",
6331 "article_id",
6332 "article_batch",
6333 "article_title",
6334 "difficulty_level",
6335 "question",
6336 "question_preview",
6337 "repeated_reading_trial",
6338 "selected_answer",
6339 "is_correct",
6340 "genre",
6341 "session",
6342 "is_practice",
6343 "unique_paragraph_id",
6344 "paragraph_id",
6345 "unique_text_id",
6346 "text_id",
6347 SOURCE_FILE_COLUMN,
6348)
6351def trial_identity_key(frame: pd.DataFrame) -> list[str]:
6352 """The columns that identify one *reading* in ``frame``.
6354 ``screen_id`` joins the key when the frame is multipart, because a screen is
6355 a coordinate space of its own and its ids restart.
6356 """
6357 key = [c for c in ("participant_id", "trial_id") if c in frame.columns]
6358 if SCREEN_ID in frame.columns:
6359 key.append(SCREEN_ID)
6360 return key
6363def _sample_trials(
6364 words: pd.DataFrame, fixations: pd.DataFrame, limit: int | None, seed: int
6365) -> tuple[pd.DataFrame, pd.DataFrame, int | None]:
6366 """Narrow both frames to ``limit`` trials, or leave them alone (VAL-7).
6368 Samples the *trial keys* rather than rows, so a chosen trial arrives whole —
6369 a half-read trial would look exactly like the two-readings-under-one-id
6370 defect this screens for. Returns the frames plus the corpus size sampled
6371 from, or ``None`` when no sampling happened.
6372 """
6373 if not limit or limit <= 0:
6374 return words, fixations, None
6375 keys = trial_keys(words) | trial_keys(fixations)
6376 if len(keys) <= limit:
6377 return words, fixations, None
6378 rng = np.random.default_rng(seed)
6379 ordered = sorted(keys) # a set's order is not stable across processes
6380 picked = {ordered[i] for i in rng.choice(len(ordered), size=limit, replace=False)}
6381 kept_words, kept_fixations = filter_to_keys(words, fixations, picked)
6382 return kept_words, kept_fixations, len(keys)
6385#: Trials VAL-7's identity check examines by default before it starts sampling.
6386#: Measured on the full OneStop corpus (24,046 trials, 5.0 M rows): the census
6387#: takes 4.23 s on every dataset load, for one warning line, and this sample
6388#: takes 1.15 s. Most of what remains is fixed — narrowing two multi-million-row
6389#: frames costs ~0.8 s whatever the sample size — which is why 500 trials only
6390#: reaches 0.84 s and is not worth the worse sample. It is a screen, not a
6391#: census: a Trial ID that under-specifies does so systematically, across every
6392#: reading it merges, so a fair sample of this size finds it.
6393TRIAL_IDENTITY_SAMPLE = 2000
6396def diagnose_trial_identity(
6397 words: pd.DataFrame,
6398 fixations: pd.DataFrame,
6399 *,
6400 sample_trials: int | None = None,
6401 seed: int = 0,
6402) -> dict[str, object]:
6403 """Report evidence that one ``trial_id`` covers more than one reading (VAL-7).
6405 ``sample_trials`` caps how many trials are examined; ``None`` scans them all
6406 (what the 🗂️ Data page's *Check every trial* button asks for). The sample is
6407 drawn with a fixed ``seed``, because a screen whose verdict flickers between
6408 reruns is worse than no screen. ``sampled_from`` reports the corpus size the
6409 sample was drawn from, or ``None`` when nothing was sampled — every count in
6410 the report is then "out of ``trials``", which is what was actually looked at.
6412 Returns a dict with:
6414 - ``duplicate_word_rows`` — rows sharing ``(key…, word_id)`` in the words
6415 table. The only *structural* signal: a word box is a property of the
6416 stimulus, so one row per word per reading is an invariant, not a
6417 heuristic — and it needs no clock.
6418 - ``repeated_fixation_id_trials`` / ``backwards_clock_trials`` — the
6419 independent cross-check, and the one that says *what* merged: a clock
6420 jumping backwards mid-trial is a second recording starting, not a
6421 regression.
6422 - ``multi_valued_columns`` — ``column → number of trials in which it takes
6423 more than one value``, over :data:`_TRIAL_WITNESS_COLUMNS`. The most
6424 diagnostic, because the column name is the remedy.
6425 - ``trials`` / ``affected_trials`` — the denominator, and how many readings
6426 any signal implicates.
6428 Pure and read-only: this reports, it never repairs. Empty frames, or frames
6429 with no id columns, produce an all-clear rather than an error.
6430 """
6431 report: dict[str, object] = {
6432 "trials": 0,
6433 "affected_trials": 0,
6434 "duplicate_word_rows": 0,
6435 "repeated_fixation_id_trials": 0,
6436 "backwards_clock_trials": 0,
6437 "multi_valued_columns": {},
6438 "mixed_source_shapes": False,
6439 "sampled_from": None,
6440 }
6441 key_frame = fixations if fixations is not None and not fixations.empty else words
6442 if key_frame is None or key_frame.empty:
6443 return report
6444 key = trial_identity_key(key_frame)
6445 if len(key) < 2: # need at least participant + trial to speak of a reading
6446 return report
6447 words, fixations, sampled_from = _sample_trials(
6448 words, fixations, sample_trials, seed
6449 )
6450 report["sampled_from"] = sampled_from
6451 # The denominator is the union across both frames, not one of them: a corpus
6452 # can carry word boxes for readers who have no fixations (the bundled demo
6453 # does), and counting the numerator over words against a fixations-only
6454 # denominator reads as "6 of 4 trials".
6455 all_keys: set = set()
6456 for frame in (words, fixations):
6457 if frame is None or frame.empty:
6458 continue
6459 fkey = trial_identity_key(frame)
6460 if len(fkey) < 2:
6461 continue
6462 sub = frame[fkey].drop_duplicates()
6463 all_keys |= {tuple(str(v) for v in row) for row in sub.to_numpy()}
6464 report["trials"] = len(all_keys)
6465 flagged: set = set()
6467 def _keys_of(frame: pd.DataFrame, mask) -> set:
6468 sub = frame.loc[mask, key].drop_duplicates()
6469 return {tuple(str(v) for v in row) for row in sub.to_numpy()}
6471 # (1) Structural: one word row per word per reading.
6472 if words is not None and not words.empty and "word_id" in words.columns:
6473 wkey = trial_identity_key(words)
6474 if len(wkey) >= 2:
6475 dup = words.duplicated(subset=wkey + ["word_id"], keep=False)
6476 report["duplicate_word_rows"] = int(dup.sum())
6477 flagged |= _keys_of(words, dup)
6479 if fixations is not None and not fixations.empty:
6480 fkey = trial_identity_key(fixations)
6481 # (2) A fixation id repeating inside one reading.
6482 if len(fkey) >= 2 and "fixation_id" in fixations.columns:
6483 counts = fixations.groupby(fkey, dropna=False)["fixation_id"].agg(
6484 lambda s: int(s.size - s.nunique())
6485 )
6486 bad = counts[counts > 0]
6487 report["repeated_fixation_id_trials"] = len(bad)
6488 flagged |= {tuple(str(v) for v in _as_tuple(k)) for k in bad.index}
6489 # (3) A clock that runs backwards mid-reading.
6490 if len(fkey) >= 2 and "timestamp_ms" in fixations.columns:
6491 clock = pd.to_numeric(fixations["timestamp_ms"], errors="coerce")
6492 ordered = pd.DataFrame({"_t": clock.to_numpy()})
6493 for col in fkey:
6494 ordered[col] = fixations[col].astype(str).to_numpy()
6495 drops = ordered.groupby(fkey, dropna=False)["_t"].apply(
6496 lambda s: bool((s.diff() < 0).any())
6497 )
6498 bad_clock = drops[drops]
6499 report["backwards_clock_trials"] = len(bad_clock)
6500 flagged |= {tuple(str(v) for v in _as_tuple(k)) for k in bad_clock.index}
6502 # (4) A column that should be constant taking several values.
6503 multi: dict[str, int] = {}
6504 for frame in (words, fixations):
6505 if frame is None or frame.empty:
6506 continue
6507 fkey = trial_identity_key(frame)
6508 if len(fkey) < 2:
6509 continue
6510 present = [c for c in _TRIAL_WITNESS_COLUMNS if c in frame.columns]
6511 if not present:
6512 continue
6513 nunique = frame.groupby(fkey, dropna=False)[present].nunique(dropna=True)
6514 for col in present:
6515 offenders = nunique[col] > 1
6516 count = int(offenders.sum())
6517 if count:
6518 multi[col] = max(multi.get(col, 0), count)
6519 flagged |= {
6520 tuple(str(v) for v in _as_tuple(k))
6521 for k in nunique.index[offenders.to_numpy()]
6522 }
6523 report["multi_valued_columns"] = dict(
6524 sorted(multi.items(), key=lambda kv: (-kv[1], kv[0]))
6525 )
6526 # #374 F3: when `source_file` is what varies, the files may be two kinds of
6527 # table joined into one (a fixation report and an interest-area report) —
6528 # adding `source_file` to the Trial ID would only hide that.
6529 report["mixed_source_shapes"] = SOURCE_FILE_COLUMN in multi and any(
6530 _mixed_source_shapes(frame) for frame in (words, fixations)
6531 )
6532 report["affected_trials"] = len(flagged)
6533 return report
6536def _mixed_source_shapes(frame: pd.DataFrame | None) -> bool:
6537 """Whether the files behind ``frame`` filled different columns (#374 F3).
6539 Each ``source_file``'s rows are reduced to the set of columns they hold any
6540 value in; two files of one export share that set, while a fixation report
6541 and an interest-area report joined into one table do not.
6542 """
6543 if frame is None or frame.empty or SOURCE_FILE_COLUMN not in frame.columns:
6544 return False
6545 filled = frame.notna().groupby(frame[SOURCE_FILE_COLUMN], sort=False).any()
6546 return len(filled.drop_duplicates()) > 1
6549def _as_tuple(key) -> tuple:
6550 """A groupby index label as a tuple, whether or not it was a MultiIndex."""
6551 return key if isinstance(key, tuple) else (key,)
6554def trial_identity_warning(report: dict[str, object]) -> str | None:
6555 """One line naming the problem and the remedy, or ``None`` when all clear.
6557 The column name is the fix, so it leads whenever there is one — "add it to
6558 the Trial ID mapping" is something the user can act on, where "N trials look
6559 wrong" is not.
6560 """
6561 if not report or not report.get("affected_trials"):
6562 return None
6563 affected = int(report["affected_trials"])
6564 total = int(report.get("trials") or 0)
6565 sampled_from = report.get("sampled_from")
6566 scope = (
6567 f"of {total:,} sampled trials (out of {int(sampled_from):,})"
6568 if sampled_from
6569 else f"of {total:,} trials"
6570 )
6571 lead = (
6572 f"**{affected:,} {scope} look like more than one trial "
6573 f"under the current Trial ID.**"
6574 )
6575 multi = report.get("multi_valued_columns") or {}
6576 if report.get("mixed_source_shapes"):
6577 return (
6578 f"{lead} Their rows come from files with different columns — one "
6579 "table probably holds both fixation and interest-area reports. "
6580 "Upload each report in its own row: Fixations, or Words (interest "
6581 "areas)."
6582 )
6583 if multi:
6584 col, count = next(iter(multi.items()))
6585 return (
6586 f"{lead} `{col}` takes more than one value inside a trial "
6587 f"({plural(count, 'trial')}) — adding it to the Trial ID mapping "
6588 "would separate them."
6589 )
6590 parts = []
6591 if report.get("duplicate_word_rows"):
6592 parts.append(plural(report["duplicate_word_rows"], "duplicated word row"))
6593 if report.get("repeated_fixation_id_trials"):
6594 parts.append(
6595 f"{plural(report['repeated_fixation_id_trials'], 'trial')} with a "
6596 "repeated fixation ID"
6597 )
6598 if report.get("backwards_clock_trials"):
6599 parts.append(
6600 f"{plural(report['backwards_clock_trials'], 'trial')} whose fixation "
6601 "onsets run backwards"
6602 )
6603 return f"{lead} Evidence: {', '.join(parts)}."
6606def count_trials(words: pd.DataFrame, fixations: pd.DataFrame) -> int:
6607 """How many distinct ``(participant_id, trial_id)`` trials the frames hold.
6609 Counts across both frames, so a words-only or fixations-only dataset is
6610 measured just as well as a paired one.
6611 """
6612 keys: set = set()
6613 for df in (words, fixations):
6614 if df is None or df.empty:
6615 continue
6616 if "participant_id" not in df.columns or "trial_id" not in df.columns:
6617 continue
6618 keys.update(zip(df["participant_id"].astype(str), df["trial_id"].astype(str)))
6619 return len(keys)
6622def diagnose_filters(
6623 words: pd.DataFrame,
6624 fixations: pd.DataFrame,
6625 steps: Sequence[tuple],
6626) -> list[dict]:
6627 """Attribute an empty trial pool to the filter(s) that caused it (UX-7).
6629 ``steps`` is ``(label, apply)`` or ``(label, apply, keys)``, where
6630 ``apply(words, fixations)`` returns the frames with *only that one* filter
6631 applied and ``keys`` is the session-state key(s) that filter is stored under
6632 (so the caller can offer "clear just this one"). Each step is measured against
6633 the **unfiltered** frames, so the result says what each filter does on its
6634 own — which is the question a user staring at an empty plot is asking. A step
6635 that alone leaves nothing is the culprit; if every step leaves something but
6636 the combination doesn't, it's their intersection, and the caller can say so.
6638 Returns one dict per step: ``{"label", "kept", "dropped", "empties", "keys"}``.
6639 """
6640 total = count_trials(words, fixations)
6641 report: list[dict] = []
6642 for step in steps:
6643 label, apply = step[0], step[1]
6644 keys = tuple(step[2]) if len(step) > 2 else ()
6645 w, f = apply(words, fixations)
6646 kept = count_trials(w, f)
6647 report.append(
6648 {
6649 "label": label,
6650 "kept": kept,
6651 "dropped": total - kept,
6652 "empties": total > 0 and kept == 0,
6653 "keys": keys,
6654 }
6655 )
6656 return report
6659def filter_raw_gaze(
6660 raw_gaze: pd.DataFrame,
6661 participants: list,
6662 trials: list,
6663) -> pd.DataFrame:
6664 """Filter raw gaze data by participants and trials."""
6665 if raw_gaze.empty:
6666 return raw_gaze
6667 mask = raw_gaze["participant_id"].isin(participants) & raw_gaze["trial_id"].isin(
6668 trials
6669 )
6670 return raw_gaze[mask]
6673def compute_canvas_size(
6674 words: pd.DataFrame, fixations: pd.DataFrame
6675) -> tuple[int, int]:
6676 """Estimate canvas size from word boxes and fixation extents.
6678 Returns the smallest power-of-100 dimensions that comfortably enclose the
6679 rightmost/bottommost data point. Falls back to DEFAULT_FIGURE_SIZE when
6680 nothing is available.
6681 """
6682 default_w, default_h = DEFAULT_FIGURE_SIZE
6683 x_candidates: list[float] = []
6684 y_candidates: list[float] = []
6686 def extent(frame: pd.DataFrame, position: str, size: str | None = None) -> float:
6687 # Coerced (BUG-54): the wizard estimates from the *raw* upload, where a
6688 # column that merely happens to be named `x` can be text — a
6689 # decimal-comma export, a unit suffix — and `float(max())` raised.
6690 if position not in frame.columns:
6691 return np.nan
6692 value = _to_number(frame[position])
6693 if size is not None and size in frame.columns:
6694 value = value + _to_number(frame[size])
6695 value = value.astype(float)
6696 return float(value[np.isfinite(value)].max())
6698 if words is not None and not words.empty and "x" in words.columns:
6699 x_candidates.append(extent(words, "x", "width"))
6700 y_candidates.append(extent(words, "y", "height"))
6701 if fixations is not None and not fixations.empty and "x" in fixations.columns:
6702 x_candidates.append(extent(fixations, "x"))
6703 y_candidates.append(extent(fixations, "y"))
6704 # NaN maxima happen when fixations ship without coordinates (AOI-sequence
6705 # data) and no word boxes were available to fill them in.
6706 x_candidates = [v for v in x_candidates if np.isfinite(v)]
6707 y_candidates = [v for v in y_candidates if np.isfinite(v)]
6708 if not x_candidates or not y_candidates:
6709 return max(int(default_w), 100), max(int(default_h), 100)
6710 width = int(np.ceil(max(x_candidates) / 100.0) * 100)
6711 height = int(np.ceil(max(y_candidates) / 100.0) * 100)
6712 return max(width, 100), max(height, 100)
6715def canvas_geometry_frames(
6716 words: pd.DataFrame | None,
6717 word_schema: dict | None,
6718 fixations: pd.DataFrame | None,
6719 fixation_schema: dict | None,
6720) -> tuple[pd.DataFrame, pd.DataFrame]:
6721 """The mapped geometry of *raw* tables, in the canonical columns
6722 :func:`compute_canvas_size` reads (DATA-46).
6724 The add-dataset wizard asks for the screen before anything is normalized, so
6725 it only has the upload as read — ``IA_LEFT`` and ``CURRENT_FIX_X``, not
6726 ``x``. Handed straight to :func:`compute_canvas_size`, an EyeLink export has
6727 no column called ``x``, and the "estimate" was the default screen under
6728 another name. This projects just the mapped coordinate columns (word boxes
6729 as edges *or* origin + size, fixation x/y) onto ``x``/``y``/``width``/
6730 ``height`` — cheap, and correct for any mapping the user has picked so far.
6731 A field that is not mapped yet is simply absent.
6732 """
6734 def column(frame: pd.DataFrame, schema: dict, key: str):
6735 name = schema.get(key)
6736 if not isinstance(name, str) or name not in frame.columns:
6737 return None
6738 return _to_number(frame[name])
6740 word_geometry = pd.DataFrame()
6741 if words is not None and not words.empty and word_schema:
6742 left, right = (
6743 column(words, word_schema, "left"),
6744 column(words, word_schema, "right"),
6745 )
6746 top, bottom = (
6747 column(words, word_schema, "top"),
6748 column(words, word_schema, "bottom"),
6749 )
6750 if left is not None and right is not None:
6751 word_geometry["x"], word_geometry["width"] = left, right - left
6752 elif (x := column(words, word_schema, "x")) is not None:
6753 word_geometry["x"] = x
6754 if (width := column(words, word_schema, "width")) is not None:
6755 word_geometry["width"] = width
6756 if top is not None and bottom is not None:
6757 word_geometry["y"], word_geometry["height"] = top, bottom - top
6758 elif (y := column(words, word_schema, "y")) is not None:
6759 word_geometry["y"] = y
6760 if (height := column(words, word_schema, "height")) is not None:
6761 word_geometry["height"] = height
6763 fixation_geometry = pd.DataFrame()
6764 if fixations is not None and not fixations.empty and fixation_schema:
6765 x, y = (
6766 column(fixations, fixation_schema, "x"),
6767 column(fixations, fixation_schema, "y"),
6768 )
6769 if x is not None and y is not None:
6770 fixation_geometry["x"], fixation_geometry["y"] = x, y
6771 return word_geometry, fixation_geometry
6774# Primary EyeLink IA measures. When a words frame already carries all of these
6775# (a pre-aggregated export, e.g. OneStop), the fixation-based recompute is a
6776# fallback whose output is discarded by the "existing values win" merge — so we
6777# skip it entirely. See compute_per_word_measures for the precedence rule.
6778_PREAGGREGATED_METRIC_COLUMNS = [
6779 "first_fixation_ms",
6780 "first_pass_gaze_duration_ms",
6781 "total_fixation_duration_ms",
6782 "n_fixations",
6783]
6786def compute_word_metrics(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame:
6787 """Return per-word reading measures.
6789 If the words table already carries pre-aggregated measures (EyeLink IA
6790 export), those values are preserved. Anything missing is computed from
6791 fixations + bounding boxes via `measures.compute_per_word_measures`.
6793 Cached on a cheap content *fingerprint* of the inputs (see
6794 ``frame_fingerprint``) rather than a full DataFrame hash, so a rerun that
6795 doesn't change the data reuses the result without re-hashing millions of
6796 rows. The frames themselves are passed un-hashed (underscore args).
6797 """
6798 return _compute_word_metrics_cached(
6799 words,
6800 fixations,
6801 cache_key=(frame_fingerprint(words), frame_fingerprint(fixations)),
6802 )
6805def preprocess_fixation_stage(
6806 words: pd.DataFrame, fixations: pd.DataFrame, settings: dict
6807) -> tuple[pd.DataFrame, pd.DataFrame]:
6808 """Cached PRE-1 stage; disabled returns the original fixation object."""
6809 if not settings.get("enabled"):
6810 return fixations, pd.DataFrame()
6811 key = (
6812 frame_fingerprint(words),
6813 frame_fingerprint(fixations),
6814 tuple(sorted(settings.items())),
6815 )
6816 result = _preprocess_fixation_stage_cached(words, fixations, settings, key)
6817 # BUG-103: a fresh copy out of the cache each rerun, named by its inputs.
6818 assign_derived(result, "preprocess_fixation_stage", (words, fixations), settings)
6819 return result
6822#: Ceiling on the caches keyed by a *single trial's* frames (PERF-6). Without
6823#: one, a bulk export over OneStop's 20,000 trials leaves 20,000 result frames
6824#: behind it — each computed once and never asked for again. Large enough that
6825#: stepping back and forth through a few dozen trials still hits the cache,
6826#: small enough that an export's footprint is the corpus, not the corpus plus a
6827#: copy of every trial's measures.
6828PER_TRIAL_CACHE_ENTRIES = 128
6831@st.cache_data(
6832 show_spinner="Preprocessing fixations…", max_entries=PER_TRIAL_CACHE_ENTRIES
6833)
6834def _preprocess_fixation_stage_cached(
6835 _words: pd.DataFrame, _fixations: pd.DataFrame, settings: dict, cache_key
6836) -> tuple[pd.DataFrame, pd.DataFrame]:
6837 from .measures import assign_fixations_to_words, enrich_fixations
6838 from .preprocessing import preprocess_fixations
6840 assigned = enrich_fixations(assign_fixations_to_words(_fixations, _words), _words)
6841 return preprocess_fixations(assigned, _words, settings=settings)
6844@st.cache_data(
6845 show_spinner="Computing reading measures…", max_entries=PER_TRIAL_CACHE_ENTRIES
6846)
6847def _compute_word_metrics_cached(
6848 _words: pd.DataFrame, _fixations: pd.DataFrame, cache_key
6849) -> pd.DataFrame:
6850 from .measures import compute_per_word_measures
6852 if _words.empty:
6853 return _words.copy()
6855 # Existing IA measures still win column-by-column inside the measure
6856 # function, but PRE-4 adds measures EyeLink exports do not usually carry.
6857 # Compute whenever fixations exist so those missing fields are not silently
6858 # absent merely because the four legacy headline columns were pre-aggregated.
6859 enriched = (
6860 compute_per_word_measures(_fixations, _words)
6861 if not _fixations.empty
6862 else _words
6863 )
6865 metric_fields = [
6866 "first_fixation_ms",
6867 "first_pass_gaze_duration_ms",
6868 "regression_path_duration_ms",
6869 "total_fixation_duration_ms",
6870 "higher_pass_fixation_duration_ms",
6871 "last_run_dwell_time_ms",
6872 "n_fixations",
6873 "skip_flag",
6874 "regression_in_count",
6875 "regression_out_count",
6876 "regression_in_flag",
6877 "regression_out_flag",
6878 "trial_dwell_time_ms",
6879 "trial_fixation_count",
6880 "trial_ia_count",
6881 "word_length",
6882 "word_length_no_punctuation",
6883 "gaze_duration_ms",
6884 "initial_landing_position",
6885 "initial_landing_distance",
6886 "number_of_regressions_in",
6887 "second_pass_duration_ms",
6888 "single_fixation_duration_ms",
6889 "first_fix_x",
6890 "first_fix_y",
6891 "gpt2_surprisal",
6892 "wordfreq_frequency",
6893 "subtlex_frequency",
6894 "universal_pos",
6895 "ptb_pos",
6896 "reduced_pos",
6897 "dependency_relation",
6898 "morphological_features",
6899 "entity_type",
6900 "head_word_index",
6901 "distance_to_head",
6902 "left_dependents_count",
6903 "right_dependents_count",
6904 ]
6905 base_fields = [
6906 "participant_id",
6907 "trial_id",
6908 "text_id",
6909 "word_id",
6910 "text",
6911 "line_idx",
6912 ]
6913 present_fields = [
6914 col for col in base_fields + metric_fields if col in enriched.columns
6915 ]
6916 metrics = enriched[present_fields].copy()
6918 numeric_fields = [
6919 "first_fixation_ms",
6920 "first_pass_gaze_duration_ms",
6921 "regression_path_duration_ms",
6922 "total_fixation_duration_ms",
6923 "higher_pass_fixation_duration_ms",
6924 "last_run_dwell_time_ms",
6925 "trial_dwell_time_ms",
6926 "trial_fixation_count",
6927 "trial_ia_count",
6928 "regression_in_count",
6929 "regression_out_count",
6930 "word_length",
6931 "word_length_no_punctuation",
6932 "gaze_duration_ms",
6933 "first_fix_x",
6934 "first_fix_y",
6935 "gpt2_surprisal",
6936 "wordfreq_frequency",
6937 "subtlex_frequency",
6938 "head_word_index",
6939 "distance_to_head",
6940 "left_dependents_count",
6941 "right_dependents_count",
6942 ]
6943 for col in numeric_fields:
6944 if col in metrics.columns:
6945 metrics[col] = pd.to_numeric(metrics[col], errors="coerce")
6946 if "n_fixations" in metrics.columns:
6947 metrics["n_fixations"] = (
6948 pd.to_numeric(metrics["n_fixations"], errors="coerce")
6949 .fillna(0)
6950 .astype("Int64")
6951 )
6952 for col in ["skip_flag", "regression_in_flag", "regression_out_flag"]:
6953 if col in metrics.columns:
6954 # BUG-7: same sentinel-aware coercion as normalization — a frame can
6955 # reach here carrying raw `'0'` / `'.'` strings (a pre-computed IA
6956 # measure joined straight in), and a truthiness cast would flag every
6957 # row True.
6958 metrics[col] = coerce_flag(metrics[col])
6959 return metrics
6962def default_filters(words: pd.DataFrame, fixations: pd.DataFrame) -> dict:
6963 """Default ("everything selected") filter dict for the current frames.
6965 Cached on a cheap content fingerprint so the full-column ``unique()`` scans
6966 don't re-run on every rerun when the data hasn't changed.
6967 """
6968 return _default_filters_cached(
6969 words,
6970 fixations,
6971 cache_key=(frame_fingerprint(words), frame_fingerprint(fixations)),
6972 )
6975@st.cache_data(show_spinner=False)
6976def _default_filters_cached(
6977 _words: pd.DataFrame, _fixations: pd.DataFrame, cache_key
6978) -> dict:
6979 # UX-166: keyed on the filtered pair, so this misses on every filter change
6980 # while everything upstream hits — the report shows the gated dataset card
6981 # while the new pool is worked out, not only once the trial list builds.
6982 progress.report()
6983 filters = dict(
6984 participants=_union_column_values(_words, _fixations, "participant_id"),
6985 trials=_union_column_values(_words, _fixations, "trial_id"),
6986 # The participant/trial lists above are the *full* unique set of the
6987 # (already trial-filtered) frame, so filter_data's membership masks are
6988 # no-ops — flag that so it can skip the two O(n) scans.
6989 _participants_cover_all=True,
6990 _trials_cover_all=True,
6991 )
6992 if "pass_index" in _fixations.columns:
6993 filters["pass_indices"] = sorted(_fixations["pass_index"].dropna().unique())
6994 if "saccade_type" in _fixations.columns:
6995 filters["saccade_types"] = sorted(
6996 _fixations["saccade_type"].dropna().astype(str).unique()
6997 )
6998 if "eye" in _fixations.columns:
6999 filters["eyes"] = sorted(_fixations["eye"].dropna().astype(str).unique())
7000 return filters