Coverage for scanpath_studio/utils.py: 96%
649 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Utility functions for trial selection, statistics, and labelling."""
3from __future__ import annotations
5import re
6from collections.abc import Callable, Iterable
8import numpy as np
9import pandas as pd
10import streamlit as st
12from . import progress
13from .annotations import current_dataset as annotations_dataset
14from .annotations import get_entry, store_for_prefix
15from .column_names import COMPUTED_SUFFIX, active_all
16from .constants import (
17 SELECTOR_ROW_GRID,
18 SELECTOR_ROW_TRIO,
19 SELECTOR_SCREEN_TRACK,
20 spoken,
21)
22from .data import frame_fingerprint, stable_id
23from .fields import labeled
24from .styles import widen_menu
26# Annotation markers shown beside a trial in the pickers (UX-6). Independent of
27# the same-text/same-participant markers (UX-4) and of each other — a trial can
28# carry any combination, so they compose. ★ favorite · 🏷️ tagged · 📝 noted.
29FAVORITE_MARKER = "★"
30TAGGED_MARKER = "🏷️"
31NOTE_MARKER = "📝"
34def annotation_markers(participant_id, trial_id, *, store=None) -> str:
35 """Composable annotation markers (★ favorite · 🏷️ tagged · 📝 noted) for a
36 trial, or ``""`` when it carries no annotations. Reads the session store —
37 the open dataset's — or ``store``, another dataset's (DATA-48)."""
38 if participant_id is None or trial_id is None:
39 return ""
40 entry = (
41 get_entry(str(participant_id), str(trial_id))
42 if store is None
43 else store.get((str(participant_id), str(trial_id))) or {}
44 )
45 marks = ""
46 if entry.get("star"):
47 marks += FAVORITE_MARKER
48 if entry.get("tags"):
49 marks += TAGGED_MARKER
50 if str(entry.get("note") or "").strip():
51 marks += NOTE_MARKER
52 return marks
55# -----------------------------------------------------------------------------
56# Trial combo building
57# -----------------------------------------------------------------------------
59#: The identity columns `build_combo_options` reads off a frame (besides any
60#: composite-trial components) — all `combo_source` has to carry.
61_COMBO_ID_COLUMNS = (
62 "participant_id",
63 "trial_id",
64 "unique_trial_id",
65 "unique_text_id",
66 "text_id",
67 "unique_paragraph_id",
68 "paragraph_id",
69 "TRIAL_INDEX",
70 "trial_index",
71)
74def combo_source(
75 fixations: pd.DataFrame,
76 words: pd.DataFrame,
77 raw_gaze: pd.DataFrame | None = None,
78) -> pd.DataFrame:
79 """The frame the trial picker's combos are built from.
81 Fixations when there are any, else words (a words-only dataset), else raw
82 gaze (a raw-gaze-only one) — and, since VIZ-45, **plus the trials only the
83 raw gaze has**. A trial recorded as samples alone is a trial: in a dataset
84 whose fixations cover other trials, or whose words table covers other
85 texts, it used to be unpickable because the picker listed the first
86 non-empty table and nothing else.
88 Returns the chosen frame itself whenever the raw gaze adds no trial (the
89 common case, and every dataset without raw gaze), so `build_combo_options`
90 keys its cache on the same object as before. Otherwise it returns a small
91 frame of identity rows — the chosen frame's, in their order, then the
92 raw-gaze-only trials' — which is all `build_combo_options` reads.
93 """
94 for primary in (fixations, words):
95 if primary is not None and not primary.empty:
96 break
97 else:
98 return raw_gaze if raw_gaze is not None else pd.DataFrame()
99 if raw_gaze is None or raw_gaze.empty:
100 return primary
101 composite_cols = tuple(st.session_state.get("_composite_trial_columns") or [])
102 combined = _combo_source_with_raw_gaze(
103 primary,
104 raw_gaze,
105 composite_cols,
106 cache_key=(frame_fingerprint(primary), frame_fingerprint(raw_gaze)),
107 )
108 return primary if combined is None else combined
111@st.cache_data(show_spinner=False, max_entries=16)
112def _combo_source_with_raw_gaze(
113 _primary: pd.DataFrame,
114 _raw_gaze: pd.DataFrame,
115 composite_cols: tuple[str, ...],
116 cache_key,
117) -> pd.DataFrame | None:
118 """`combo_source`'s identity rows, or ``None`` when raw gaze adds no trial."""
119 progress.report()
120 from .data import trial_keys
122 # One deduplication per table, on the identity columns only; every key
123 # set below comes off those small frames rather than another pass over
124 # every sample (PERF: three full scans at 5M samples was ~0.8 s a miss).
125 wanted = [*_COMBO_ID_COLUMNS, *composite_cols]
126 primary_cols = [c for c in dict.fromkeys(wanted) if c in _primary.columns]
127 rows = _primary[primary_cols].drop_duplicates()
128 raw_cols = [c for c in dict.fromkeys(wanted) if c in _raw_gaze.columns]
129 raw_rows = _raw_gaze[raw_cols].drop_duplicates()
130 extra_keys = trial_keys(raw_rows) - trial_keys(rows)
131 if not extra_keys:
132 return None
133 index = pd.MultiIndex.from_arrays(
134 [raw_rows["participant_id"].astype(str), raw_rows["trial_id"].astype(str)]
135 )
136 raw_rows = raw_rows[index.isin(extra_keys)].copy()
137 # The picker keys on the primary frame's trial and text columns; a raw-gaze
138 # row that lacks one takes its own trial id / text id, which is what those
139 # columns mean for a normalized frame (`data.normalize_raw_gaze`).
140 if "unique_trial_id" in rows.columns and "unique_trial_id" not in raw_rows:
141 raw_rows["unique_trial_id"] = raw_rows["trial_id"]
142 for text_col in ("unique_text_id", "text_id", "unique_paragraph_id"):
143 if text_col in rows.columns and text_col not in raw_rows.columns:
144 raw_rows[text_col] = raw_rows.get("text_id", raw_rows["trial_id"])
145 return pd.concat([rows, raw_rows], ignore_index=True)
148def build_combo_options(
149 fixations: pd.DataFrame,
150) -> tuple[pd.DataFrame, list[str], dict[str, tuple[str, str]]]:
151 """Build participant/trial/text combinations for selection UI.
153 Returns:
154 Tuple of (combos DataFrame, label list, label-to-combo mapping).
156 Cached on a cheap fingerprint of the frame + the composite-trial columns, so
157 the full-frame ``drop_duplicates`` + label build don't re-run on every rerun
158 (e.g. selecting a different trial). The session-state read happens here, in
159 the un-cached wrapper, and is threaded into the cached core as an argument.
160 """
161 composite_cols = tuple(st.session_state.get("_composite_trial_columns") or [])
162 return build_combo_options_for(fixations, composite_cols)
165def build_combo_options_for(
166 fixations: pd.DataFrame,
167 composite_cols: tuple[str, ...] = (),
168) -> tuple[pd.DataFrame, list[str], dict[str, tuple[str, str]]]:
169 """`build_combo_options` for a frame that is **not** the active dataset.
171 CMP-8 §2: a comparison source has its own composite-trial columns, so the
172 session-state read in `build_combo_options` would answer for the wrong
173 dataset. Everything else — the cache key, the cached core — is shared.
174 """
175 composite_cols = tuple(composite_cols or ())
176 return _build_combo_options_cached(
177 fixations,
178 composite_cols,
179 cache_key=(frame_fingerprint(fixations), composite_cols),
180 )
183# UX-166: the dataset card lists this step.
184@st.cache_data(show_spinner=False)
185def _build_combo_options_cached(
186 _fixations: pd.DataFrame,
187 composite_cols: tuple[str, ...],
188 cache_key,
189) -> tuple[pd.DataFrame, list[str], dict[str, tuple[str, str]]]:
190 progress.report() # a miss: real work, so a gated card over it may show
191 fixations = _fixations
192 trial_col = (
193 "unique_trial_id" if "unique_trial_id" in fixations.columns else "trial_id"
194 )
195 # The text/passage column is optional. Normalized frames carry a text_id (it
196 # falls back to trial_id when no text is mapped), but a frame may arrive with
197 # only the source name (e.g. unique_paragraph_id — the pre-rename text id, and
198 # which can also be a composite-trial component). Detect it via the same
199 # priority list normalization uses, and *copy* it to text_id rather than
200 # renaming so a shared composite component column survives for the picker.
201 text_col = next(
202 (
203 c
204 for c in (
205 "unique_text_id",
206 "text_id",
207 "unique_paragraph_id",
208 "paragraph_id",
209 )
210 if c in fixations.columns
211 ),
212 None,
213 )
214 combo_cols = ["participant_id", trial_col]
215 if text_col is not None and text_col not in combo_cols:
216 combo_cols.append(text_col)
217 for col in ["unique_trial_id", "unique_text_id", "TRIAL_INDEX", "trial_index"]:
218 if col in fixations.columns and col not in combo_cols:
219 combo_cols.append(col)
220 # Carry the composite trial id's component columns through, so the trial
221 # picker can detect a composite id and cascade on identity (see select_trial).
222 for col in composite_cols:
223 if col in fixations.columns and col not in combo_cols:
224 combo_cols.append(col)
226 # UX-24: preserve each trial's first appearance before the historical
227 # participant/id sort. The visible pool may still default to Trial ID, but
228 # the ⇅ menu can now reconstruct source-file order exactly.
229 combos = fixations[combo_cols].drop_duplicates().copy()
230 combos["_data_order"] = np.arange(len(combos), dtype=int)
231 combos = combos.rename(columns={trial_col: "trial_id"})
232 if "text_id" not in combos.columns:
233 combos["text_id"] = (
234 combos[text_col] if text_col is not None else combos["trial_id"]
235 )
236 if trial_col == "unique_trial_id" and "unique_trial_id" not in combos.columns:
237 combos["unique_trial_id"] = combos["trial_id"]
238 if text_col == "unique_text_id" and "unique_text_id" not in combos.columns:
239 combos["unique_text_id"] = combos["text_id"]
240 sort_cols = ["participant_id"]
241 if "TRIAL_INDEX" in combos.columns:
242 sort_cols.append("TRIAL_INDEX")
243 elif "trial_index" in combos.columns:
244 sort_cols.append("trial_index")
245 sort_cols.append("trial_id")
246 combos = combos.sort_values(sort_cols)
248 combo_labels = [
249 f"{row.participant_id} / {row.trial_id} · {row.text_id}"
250 for row in combos.itertuples()
251 ]
252 label_to_combo = dict(
253 zip(
254 combo_labels,
255 combos[["participant_id", "trial_id"]].itertuples(index=False, name=None),
256 )
257 )
258 return combos, combo_labels, label_to_combo
261@st.cache_data(show_spinner=False)
262def _trial_positions(_frame: pd.DataFrame, cache_key) -> dict[tuple[str, str], object]:
263 """Map ``(participant_id, trial_id)`` → positional row indices.
265 Built once per frame (cached on its fingerprint) so extracting a single
266 trial is an O(trial) ``iloc`` rather than an O(corpus) boolean mask on every
267 rerun — and shared across the tabs, which all slice the same filtered frames.
268 """
269 if _frame is None or _frame.empty:
270 return {}
271 grouped = _frame.groupby(["participant_id", "trial_id"], sort=False).indices
272 # Normalise keys to (str, str) so lookups match the picker's string values.
273 return {(str(p), str(t)): idx for (p, t), idx in grouped.items()}
276def extract_trial(frame: pd.DataFrame, participant_id, trial_id) -> pd.DataFrame:
277 """Rows of one (participant, trial), sliced via the cached position index.
279 Equivalent to ``frame[(frame.participant_id == p) & (frame.trial_id == t)]``
280 but O(trial) instead of O(corpus) once the index is built — the per-rerun win
281 on large datasets, where every tab extracts the selected trial."""
282 if frame is None or getattr(frame, "empty", True):
283 return frame
284 positions = _trial_positions(frame, cache_key=frame_fingerprint(frame))
285 pos = positions.get((str(participant_id), str(trial_id)))
286 if pos is None or len(pos) == 0:
287 return frame.iloc[0:0]
288 return frame.iloc[pos]
291# -----------------------------------------------------------------------------
292# Trial selection UI
293# -----------------------------------------------------------------------------
295# UX-10 · sorting the trial pool.
296#
297# The picker listed trials in data order, so finding "the slowest reader", "the
298# one with the most fixations" or "the trials this reader got wrong" meant
299# scrolling the whole list. These build a sort key per trial from three sources:
300# computed per-trial stats, reader/text properties, and any trial-level column
301# the dataset carries. Pure and frame-driven, so they're testable without the UI.
302TRIAL_SORT_DEFAULT = "Trial ID"
303#: UX-171: the order the trials appear in the data — the picker's default when
304#: the combos carry it. ``Trial ID`` (sorted by id, so ``1, 10, 100, 2`` for
305#: numeric ids) stays in the menu as a choice.
306TRIAL_SORT_DATA_ORDER = "Data order"
307# Computed stat label → (frame it needs, how to aggregate it per trial).
308# "fixations" / "words" name which frame the aggregation runs on.
309# DATA-66: these are the app's, not columns of the dataset, so they say so and
310# are listed after the dataset's own columns.
311_TRIAL_SORT_STATS = {
312 "Fixation count (computed)": ("fixations", "size"),
313 "Total fixation time, s (computed)": ("fixations", "duration_sum_s"),
314 "Mean fixation duration, ms (computed)": ("fixations", "duration_mean"),
315 "Word count (computed)": ("words", "size"),
316 "First timestamp (computed)": ("fixations", "timestamp_min"),
317}
318# Columns worth offering as a sort key when the dataset carries them, in the
319# order they're shown. Reader properties first, then text, then behaviour.
320_TRIAL_SORT_PREFERRED_COLS = (
321 "participant_id",
322 "text_id",
323 "unique_text_id",
324 "paragraph_id",
325 "difficulty_level",
326 "question_preview",
327 "repeated_reading_trial",
328 "is_correct",
329 "genre",
330 "session",
331 "pp_age",
332 "pp_gender",
333 "TRIAL_INDEX",
334 "trial_index",
335)
337# Event/geometry fields can occasionally be constant by accident (for example a
338# one-fixation trial), but that does not make them trial metadata. Keep them out
339# of the generic metadata tail; computed timing/count keys above are the useful
340# sortable representation of those event columns.
341_TRIAL_SORT_EXCLUDED_COLS = {
342 "trial_id",
343 "unique_trial_id",
344 "word",
345 "token",
346 "text",
347 "sentence",
348 "question",
349 "answer",
350 "response",
351 "x",
352 "y",
353 "x_start",
354 "x_end",
355 "y_start",
356 "y_end",
357 "xmin",
358 "xmax",
359 "ymin",
360 "ymax",
361 "width",
362 "height",
363 "timestamp",
364 "timestamp_ms",
365 "duration",
366 "duration_ms",
367 "order_in_trial",
368 "fixation_index",
369 "word_index",
370 "word_id",
371 "ia_id",
372 "char_index",
373 "line_index",
374 "source_file",
375}
376_TRIAL_SORT_PRIVATE_NAME_PARTS = (
377 "path",
378 "filepath",
379 "filename",
380 "directory",
381 "folder",
382 "url",
383 "uri",
384)
385_TRIAL_SORT_GEOMETRY_SUFFIXES = (
386 "_x",
387 "_y",
388 "_xmin",
389 "_xmax",
390 "_ymin",
391 "_ymax",
392 "_x_start",
393 "_x_end",
394 "_y_start",
395 "_y_end",
396 "_width",
397 "_height",
398)
401def _trial_sort_column_allowed(column: object) -> bool:
402 """Whether ``column`` can be discovered as generic trial metadata."""
403 name = str(column)
404 lower = name.lower()
405 if name.startswith("_") or lower in _TRIAL_SORT_EXCLUDED_COLS:
406 return False
407 if lower.endswith(_TRIAL_SORT_GEOMETRY_SUFFIXES):
408 return False
409 return not any(part in lower for part in _TRIAL_SORT_PRIVATE_NAME_PARTS)
412def _effective_trial_field(
413 frame: pd.DataFrame | None, trial_field: str, picker_ids: set[str]
414) -> str | None:
415 """Find the frame column that names the picker's effective trial ids."""
416 if frame is None or frame.empty:
417 return None
418 for field in (trial_field, "unique_trial_id", "trial_id"):
419 if field not in frame.columns:
420 continue
421 values = set(frame[field].dropna().astype(str).unique())
422 if values & picker_ids:
423 return field
424 return None
427def _is_missing_scalar(value) -> bool:
428 try:
429 missing = pd.isna(value)
430 except (TypeError, ValueError):
431 return False
432 return bool(missing) if isinstance(missing, (bool, np.bool_)) else False
435def _same_sort_value(left, right) -> bool:
436 """Scalar equality for reconciling the words and fixations tables."""
437 if _is_missing_scalar(left) or _is_missing_scalar(right):
438 # Partial missingness is not a conflict; the table carrying a value wins.
439 return True
440 try:
441 equal = left == right
442 except (TypeError, ValueError):
443 return False
444 return bool(equal) if isinstance(equal, (bool, np.bool_)) else False
447def _looks_like_free_text(series: pd.Series) -> bool:
448 """Reject prose-like cells while retaining ordinary categorical metadata."""
449 values = [str(v).strip() for v in series if not _is_missing_scalar(v)]
450 return bool(values) and any(len(v) > 200 or len(v.split()) > 24 for v in values)
453def _trial_level_columns_from_frame(
454 frame: pd.DataFrame | None,
455 trial_field: str,
456 picker_ids: set[str],
457 participants: set[str],
458) -> dict[str, pd.Series]:
459 """Discover one scalar value per active trial directly from one source table.
461 Grouping includes participant identity, preventing repeated plain trial ids
462 from being merged before the active picker scope is applied. The returned
463 Series uses the picker's effective id because that is what
464 ``sort_trial_options`` consumes.
465 """
466 identity = _effective_trial_field(frame, trial_field, picker_ids)
467 if frame is None or frame.empty or identity is None:
468 return {}
470 scoped = frame
471 if participants and "participant_id" in scoped.columns:
472 scoped = scoped[scoped["participant_id"].astype(str).isin(participants)]
473 scoped = scoped[scoped[identity].astype(str).isin(picker_ids)]
474 if scoped.empty:
475 return {}
477 group_cols = [identity]
478 if "participant_id" in scoped.columns and identity != "participant_id":
479 group_cols.insert(0, "participant_id")
480 grouped = scoped.groupby(group_cols, sort=False, dropna=False)
481 discovered: dict[str, pd.Series] = {}
482 for col in scoped.columns:
483 if col == identity or not _trial_sort_column_allowed(col):
484 continue
485 try:
486 if (grouped[col].nunique(dropna=False) > 1).any():
487 continue
488 if col in group_cols:
489 values = scoped[group_cols].drop_duplicates().copy()
490 else:
491 values = grouped[col].agg(lambda cells: cells.iloc[0]).reset_index()
492 # If the same effective id survives for multiple participants, it is
493 # usable only when those rows agree. A participant-narrowed picker
494 # naturally has one row here; a global ambiguous picker is not
495 # allowed to choose one participant silently.
496 by_id = values.groupby(values[identity].astype(str), sort=False)[col]
497 if (by_id.nunique(dropna=False) > 1).any():
498 continue
499 deduped = values.drop_duplicates(subset=[identity])
500 except (TypeError, ValueError):
501 # Nested/list-like event payloads are not sortable scalar metadata.
502 continue
503 series = pd.Series(
504 deduped[col].to_numpy(),
505 index=deduped[identity].astype(str).to_numpy(),
506 )
507 if series.dropna().empty or _looks_like_free_text(series):
508 continue
509 discovered[str(col)] = series
510 return discovered
513def _merge_trial_level_sources(
514 sources: Iterable[dict[str, pd.Series]],
515) -> dict[str, pd.Series]:
516 """Merge compatible metadata sources; omit cross-table disagreements."""
517 by_column: dict[str, list[pd.Series]] = {}
518 for source in sources:
519 for col, series in source.items():
520 by_column.setdefault(col, []).append(series)
522 merged: dict[str, pd.Series] = {}
523 for col, series_list in by_column.items():
524 combined = pd.Series(dtype=object)
525 conflict = False
526 for series in series_list:
527 current = series.copy()
528 current.index = current.index.astype(str)
529 for trial_id in combined.index.intersection(current.index):
530 if not _same_sort_value(combined[trial_id], current[trial_id]):
531 conflict = True
532 break
533 if conflict:
534 break
535 # pandas warns when concatenation/combine_first has to infer a
536 # dtype from an empty object Series. The first real source needs no
537 # merge at all; starting from it also preserves its native dtype.
538 combined = current if combined.empty else combined.combine_first(current)
539 if not conflict and not combined.empty:
540 merged[col] = combined
541 return merged
544@st.cache_data(show_spinner=False, max_entries=16)
545def _trial_level_sort_columns_cached(
546 _combos: pd.DataFrame,
547 _words: pd.DataFrame | None,
548 _fixations: pd.DataFrame | None,
549 trial_field: str,
550 cache_key,
551) -> dict[str, pd.Series]:
552 """Cached metadata discovery over the participant-scoped picker frames."""
553 del cache_key # explicit hash input for the underscore-prefixed frames
554 if _combos is None or _combos.empty or trial_field not in _combos.columns:
555 return {}
556 picker_ids = set(_combos[trial_field].dropna().astype(str).unique())
557 participants = (
558 set(_combos["participant_id"].dropna().astype(str).unique())
559 if "participant_id" in _combos.columns
560 else set()
561 )
562 return _merge_trial_level_sources(
563 (
564 _trial_level_columns_from_frame(
565 _combos, trial_field, picker_ids, participants
566 ),
567 _trial_level_columns_from_frame(
568 _words, trial_field, picker_ids, participants
569 ),
570 _trial_level_columns_from_frame(
571 _fixations, trial_field, picker_ids, participants
572 ),
573 )
574 )
577def _trial_level_sort_columns(
578 combos: pd.DataFrame,
579 trial_field: str,
580 words: pd.DataFrame | None,
581 fixations: pd.DataFrame | None,
582) -> dict[str, pd.Series]:
583 return _trial_level_sort_columns_cached(
584 combos,
585 words,
586 fixations,
587 trial_field,
588 cache_key=(
589 frame_fingerprint(combos),
590 frame_fingerprint(words),
591 frame_fingerprint(fixations),
592 trial_field,
593 ),
594 )
597def _per_trial_stat(frame: pd.DataFrame, trial_field: str, how: str) -> pd.Series:
598 """One computed stat per trial id, as a Series indexed by that id."""
599 if frame is None or frame.empty or trial_field not in frame.columns:
600 return pd.Series(dtype=float)
601 grouped = frame.groupby(frame[trial_field].astype(str), sort=False)
602 if how == "size":
603 return grouped.size().astype(float)
604 if how == "timestamp_min":
605 if "timestamp_ms" not in frame.columns:
606 return pd.Series(dtype=float)
607 return pd.to_numeric(grouped["timestamp_ms"].min(), errors="coerce").astype(
608 float
609 )
610 if "duration_ms" not in frame.columns:
611 return pd.Series(dtype=float)
612 durations = grouped["duration_ms"].agg("sum" if "sum" in how else "mean")
613 return (durations / 1000.0) if how.endswith("_s") else durations.astype(float)
616def trial_sort_keys(
617 combos: pd.DataFrame,
618 trial_field: str,
619 *,
620 words: pd.DataFrame | None = None,
621 fixations: pd.DataFrame | None = None,
622 label_of: Callable[[str], str] = str,
623) -> dict[str, pd.Series]:
624 """Available sort keys (UX-10): label → Series indexed by trial id.
626 The dataset's own trial-level columns come first, each under ``label_of``
627 (the dataset's own name, DATA-66 — `ColumnNames.label`), then the statistics
628 Scanpath Studio computes per trial, marked as computed.
630 Offers a computed stat only when the frame it needs is present, and a column
631 only when it is actually trial-level in the active participant-scoped words,
632 fixations, or combo frame. This deliberately discovers metadata before the
633 lossy combo projection can discard it.
634 """
635 keys: dict[str, pd.Series] = {}
636 # This rank was captured before build_combo_options' canonical sort.
637 if (
638 combos is not None
639 and not combos.empty
640 and trial_field in combos.columns
641 and "_data_order" in combos.columns
642 ):
643 deduped = combos.drop_duplicates(subset=[trial_field])
644 keys[TRIAL_SORT_DATA_ORDER] = pd.Series(
645 deduped["_data_order"].to_numpy(),
646 index=deduped[trial_field].astype(str).to_numpy(),
647 )
648 has_combos = (
649 combos is not None and not combos.empty and trial_field in combos.columns
650 )
651 if has_combos:
652 discovered = _trial_level_sort_columns(combos, trial_field, words, fixations)
653 ordered_cols = [c for c in _TRIAL_SORT_PREFERRED_COLS if c in discovered]
654 ordered_cols.extend(
655 sorted(set(discovered) - set(ordered_cols), key=str.casefold)
656 )
657 labelled = [(col, label_of(col)) for col in ordered_cols if col != trial_field]
658 # The dataset's own columns first, then the ones the app made.
659 labelled.sort(key=lambda pair: pair[1].endswith(COMPUTED_SUFFIX))
660 for col, label in labelled:
661 series = discovered[col]
662 if label in keys:
663 # An alias read from the same column sorts the same way: once.
664 if keys[label].equals(series):
665 continue
666 label = f"{label} ({col})"
667 elif label in (TRIAL_SORT_DEFAULT, TRIAL_SORT_DATA_ORDER):
668 # A column the dataset itself calls "Trial ID" is not the menu's.
669 label = f"{label} ({col})"
670 keys[label] = series
671 keys.update(
672 _trial_sort_stats_cached(
673 combos if has_combos else None,
674 words,
675 fixations,
676 trial_field,
677 cache_key=(
678 frame_fingerprint(combos) if has_combos else None,
679 frame_fingerprint(words),
680 frame_fingerprint(fixations),
681 trial_field,
682 ),
683 )
684 )
685 return keys
688@st.cache_data(show_spinner=False, max_entries=16)
689def _trial_sort_stats_cached(
690 _combos: pd.DataFrame | None,
691 _words: pd.DataFrame | None,
692 _fixations: pd.DataFrame | None,
693 trial_field: str,
694 cache_key,
695) -> dict[str, pd.Series]:
696 """The computed sort keys (fixation count, reading time …), label → Series.
698 Each is a group-by over the whole fixation or word table, ~0.25 s a rerun
699 at OneStop scale for numbers that change only with the trial pool."""
700 del cache_key # explicit hash input for the underscore-prefixed frames
701 picker_ids = (
702 set(_combos[trial_field].dropna().astype(str).unique())
703 if _combos is not None
704 else set()
705 )
706 stats: dict[str, pd.Series] = {}
707 for label, (which, how) in _TRIAL_SORT_STATS.items():
708 frame = _fixations if which == "fixations" else _words
709 field = _effective_trial_field(frame, trial_field, picker_ids)
710 series = _per_trial_stat(frame, field or trial_field, how)
711 if not series.empty:
712 stats[label] = series
713 return stats
716def sort_trial_options(
717 options: list[str],
718 key_series: pd.Series | None,
719 *,
720 descending: bool = False,
721) -> list[str]:
722 """Order ``options`` (trial ids) by ``key_series``, ties broken by id.
724 Trials the key doesn't cover sort last regardless of direction — an unranked
725 trial is missing information, not an extreme value, so it shouldn't lead.
726 """
727 if key_series is None or key_series.empty:
728 return sorted(options)
729 lookup = key_series.to_dict()
730 ranked = [o for o in options if o in lookup and pd.notna(lookup[o])]
731 ranked_set = set(ranked)
732 unranked = sorted(o for o in options if o not in ranked_set)
733 ranked.sort(key=lambda o: (_sort_scalar(lookup[o]), o), reverse=descending)
734 return ranked + unranked
737def _sort_scalar(value):
738 """A comparable key for a cell that may be numeric, boolean or text."""
739 if isinstance(value, bool):
740 return (0, float(value))
741 try:
742 return (0, float(value))
743 except (TypeError, ValueError):
744 return (1, str(value))
747def format_sort_value(value) -> str:
748 """A sort key's value, short enough to ride along in a picker option.
750 Sorting the pool is only useful if you can *see* what you sorted by — an
751 ordering with the ordering key hidden just looks shuffled. Integers keep a
752 thousands separator, floats get one decimal, booleans read Yes/No.
753 """
754 if value is None or (isinstance(value, float) and pd.isna(value)):
755 return "—"
756 if isinstance(value, (bool, np.bool_)):
757 return "Yes" if value else "No"
758 if isinstance(value, (int, float, np.integer, np.floating)):
759 number = float(value)
760 return f"{number:,.0f}" if number == int(number) else f"{number:,.1f}"
761 return str(value)
764def _trial_display_label(trial_id) -> str:
765 """Human-readable label for a trial id in the pickers.
767 A per-page trial id reads cleanly — ``Lit_Alchemist_4__page_07`` →
768 ``Lit_Alchemist_4 · page 7`` (the id stays zero-padded so it sorts
769 numerically; only the display drops the padding). Any other id passes
770 through unchanged, so this is a no-op for every other corpus. MultiplEYE
771 used to be the one producing those ids; since DATA-24 its pages are screens
772 inside one trial, so this is now generic dressing for whatever ships them."""
773 text = str(trial_id)
774 stim, sep, page = text.rpartition("__page_")
775 if sep and page.isdigit():
776 return f"{stim} · page {int(page)}"
777 return text
780# --- UX-187: a trial id spelled out part by part ------------------------------
781# A trial id is usually several ids joined with "_" — OneStop's
782# `l37_1129_2_2_1_Adv_r0` is reader `l37_1129`, text `2_2_1_Adv`, first reading.
783# The pickers show it with " · " between the parts so you can tell where one
784# ends, which a plain split on "_" cannot do: the reader id has an underscore of
785# its own. So the parts are found from the ids the trial is known to be made of —
786# its participant, its text, a composite mapping's columns — and an id that
787# matches none of them is shown exactly as it is. Only the display changes: the
788# selection, the deep link and every export keep the id itself.
789#
790# UX-202: the text id is one part, as it is everywhere else in the app (it was
791# split into OneStop's batch · article · paragraph · level), and a first reading
792# (`r0`) is not shown — only a repeated one says which reading it is.
794#: What the pickers put between the parts of a trial id.
795TRIAL_ID_PART_SEPARATOR = " · "
797#: UX-202 — the reading number a first reading carries, which the display leaves
798#: out. OneStop composes `_r0` / `_r1`; `data._disambiguate_repeated_readings`
799#: leaves a first reading unsuffixed and numbers the next `_r2`.
800FIRST_READING = "r0"
802_READING_SUFFIX = re.compile(r"_(r\d+)$")
805def trial_id_parts(
806 trial_id,
807 *,
808 participant_id=None,
809 text_id=None,
810 components: list[tuple[str, str]] | None = None,
811) -> list[tuple[str, str]]:
812 """``(name, value)`` for each part of ``trial_id``, in the order it is written.
814 ``components`` (a composite trial mapping's ``(column, value)`` pairs) are the
815 answer outright. Otherwise the id is read as ``[<participant>_]<text>[_rN]``.
816 An id that shape does not account for entirely is one part,
817 ``("trial id", trial_id)``.
818 """
819 if components:
820 return [(str(name), str(value)) for name, value in components]
821 tid = str(trial_id)
822 whole = [("trial id", tid)]
823 pid = "" if participant_id is None else str(participant_id)
824 text = "" if text_id is None else str(text_id)
825 parts: list[tuple[str, str]] = []
826 rest = tid
827 if pid and rest.startswith(f"{pid}_"):
828 parts.append(("participant", pid))
829 rest = rest[len(pid) + 1 :]
830 if text and text != tid and (rest == text or rest.startswith(f"{text}_")):
831 parts.append(("text", text))
832 rest = rest[len(text) + 1 :]
833 # Split only an id its parts account for entirely: the text must be in it,
834 # and anything after the text a reading number. `synthetic_2line_demo`
835 # starts with its reader's id, but is not made of it.
836 if not parts or parts[-1][0] == "participant":
837 return whole
838 if rest:
839 if not (rest[:1] == "r" and rest[1:].isdigit()):
840 return whole
841 parts.append(("reading", rest))
842 return parts
845def shown_parts(parts: list[tuple[str, str]]) -> list[tuple[str, str]]:
846 """The parts the display writes: all but a first reading (UX-202)."""
847 return [
848 (name, value)
849 for name, value in parts
850 if not (name == "reading" and value == FIRST_READING)
851 ]
854def trial_id_display(parts: list[tuple[str, str]]) -> str:
855 """A trial id as the pickers show it: its parts, `TRIAL_ID_PART_SEPARATOR`-joined.
857 A first reading is left out (UX-202), and an id that does not split still
858 has a trailing reading number set off by the separator, not an underscore.
859 """
860 parts = shown_parts(parts)
861 if len(parts) == 1:
862 label = _trial_display_label(parts[0][1])
863 return _READING_SUFFIX.sub(rf"{TRIAL_ID_PART_SEPARATOR}\1", label)
864 return TRIAL_ID_PART_SEPARATOR.join(value for _, value in parts)
867def trial_id_layout(
868 combos: pd.DataFrame,
869 trial_field: str = "trial_id",
870 *,
871 composite_cols: Iterable[str] = (),
872) -> tuple[dict[str, str], tuple[str, ...]]:
873 """Display string per trial id in ``combos``, plus the part names they share.
875 ``composite_cols`` are the composite trial mapping's columns (carried on
876 ``combos``). The part names are the most common shown layout's — what the
877 picker's help names — with ``"reading"`` added when any trial shows one, and
878 empty when no id splits.
879 """
880 if combos.empty or trial_field not in combos.columns:
881 return {}, ()
882 composite = [c for c in composite_cols if c in combos.columns]
883 text_field = next(
884 (c for c in ("unique_text_id", "text_id") if c in combos.columns), None
885 )
886 rows = combos.drop_duplicates(subset=[trial_field])
887 trial_ids = rows[trial_field].astype(str).to_numpy()
888 pids = rows["participant_id"].to_numpy() if "participant_id" in rows else None
889 texts = rows[text_field].to_numpy() if text_field else None
890 comp_values = (
891 rows[composite].apply(stable_id).to_numpy() if len(composite) > 1 else None
892 )
893 display: dict[str, str] = {}
894 layouts: dict[tuple[str, ...], int] = {}
895 any_reading = False
896 for i, tid in enumerate(trial_ids):
897 parts = trial_id_parts(
898 tid,
899 participant_id=None if pids is None else pids[i],
900 text_id=None if texts is None else texts[i],
901 components=(
902 None
903 if comp_values is None
904 else list(zip(composite, comp_values[i], strict=True))
905 ),
906 )
907 display[tid] = trial_id_display(parts)
908 shown = shown_parts(parts)
909 if len(shown) > 1:
910 names = tuple(name for name, _ in shown if name != "reading")
911 layouts[names] = layouts.get(names, 0) + 1
912 any_reading = any_reading or any(name == "reading" for name, _ in shown)
913 names = max(layouts, key=layouts.__getitem__) if layouts else ()
914 if names and any_reading:
915 names = (*names, "reading")
916 return display, names
919def trial_id_shown(
920 trial_id,
921 *frames: pd.DataFrame | None,
922 participant_id=None,
923 composite_cols: Iterable[str] = (),
924) -> str:
925 """One trial's id as the pickers show it, read off the first of ``frames``
926 (that trial's own rows) that has any. ``participant_id`` overrides the
927 frame's — a cross-dataset B carries a namespaced one (`qualify_for_compare`)."""
928 frame = next((f for f in frames if f is not None and not f.empty), None)
929 if frame is None:
930 return trial_id_display(trial_id_parts(trial_id))
931 row = frame.iloc[:1].copy()
932 row["trial_id"] = str(trial_id)
933 if participant_id is not None:
934 row["participant_id"] = participant_id
935 display, _ = trial_id_layout(row, composite_cols=composite_cols)
936 return display.get(str(trial_id), str(trial_id))
939def trial_id_help(part_names: tuple[str, ...]) -> str:
940 """The sentence the pickers' help gives about what a trial id is made of."""
941 if not part_names:
942 return ""
943 sentence = (
944 "A trial id is written as its parts, "
945 f"**{TRIAL_ID_PART_SEPARATOR.join(part_names)}**."
946 )
947 if "reading" in part_names:
948 sentence += (
949 " A repeated reading ends in its number (`r1`, `r2` …); a first"
950 " reading has none."
951 )
952 return sentence
955def _render_trial_sort_popover(
956 host,
957 combos: pd.DataFrame,
958 trial_field: str,
959 key_prefix: str,
960 *,
961 words: pd.DataFrame | None,
962 fixations: pd.DataFrame | None,
963) -> tuple[pd.Series | None, bool, str]:
964 """The ⇅ sort control beside the trial picker (UX-10).
966 Lives in a popover rather than inline: the picker row is already a selectbox,
967 a slider and two step buttons wide, and sorting is a "set it once" choice, not
968 a per-trial one. Returns ``(key_series, descending, choice)`` for
969 :func:`sort_trial_options` — ``(None, False, TRIAL_SORT_DEFAULT)`` for the
970 default id order. The chosen key's *name* comes back too, because the picker
971 labels the ordering it's showing.
972 """
973 keys = trial_sort_keys(
974 combos,
975 trial_field,
976 words=words,
977 fixations=fixations,
978 label_of=active_all(st.session_state).label,
979 )
980 if not keys:
981 return None, False, TRIAL_SORT_DEFAULT
982 # UX-171: data order leads and is the default; Trial ID follows it.
983 default = TRIAL_SORT_DATA_ORDER if TRIAL_SORT_DATA_ORDER in keys else None
984 options = [
985 *([default] if default else []),
986 TRIAL_SORT_DEFAULT,
987 *(k for k in keys if k != default),
988 ]
989 state_key = f"{key_prefix}_trial_sort"
990 if st.session_state.get(state_key) not in options:
991 st.session_state[state_key] = options[0]
992 # UX-200: named for screen readers; `styles.py` draws ⇅ alone.
993 with host.popover(
994 "Sort the trial list",
995 width="content",
996 wrap=True,
997 help="Sort the trial list",
998 key=f"iconpop_sort_trial_{key_prefix}",
999 ):
1000 choice = labeled(
1001 st,
1002 "selectbox",
1003 "Sort trials by",
1004 options=options,
1005 key=state_key,
1006 help="Reorder the trial list by a computed statistic or by a participant, "
1007 "text or condition property.",
1008 )
1009 descending = labeled(
1010 st,
1011 "checkbox",
1012 "Descending",
1013 key=f"{key_prefix}_trial_sort_desc",
1014 help="Reverse the order.",
1015 )
1016 if choice == TRIAL_SORT_DEFAULT:
1017 return None, False, TRIAL_SORT_DEFAULT
1018 return keys[choice], bool(descending), choice
1021# --- CMP-13: one ◀ ▶ that advances both compared trials ----------------------
1022# The two pickers are built in different modules (A here, B in `tabs.py`), so the
1023# linked step is a callback on one side writing the *other* side's selection. It
1024# needs the other list, which is why each picker publishes what it just rendered.
1025# Both directions clamp independently: per the settled call, a side that has run
1026# out simply stays put while the other keeps stepping.
1028#: The ⚙️ Compare options checkbox that arms the link. UI-only — a navigation
1029#: control, not a render setting, so it is deliberately not on the share link or
1030#: in a saved config (same call as ``share_identity_mode``).
1031COMPARE_STEP_LINK_KEY = "single_compare_step_linked"
1033#: Scanpath B's canonical selection (a *label*; see the snapshot note below).
1034COMPARE_TRIAL_KEY = "single_compare_trial"
1036#: What the *Compare To* picker last rendered: ``(label, participant, trial)``
1037#: per candidate, in display order.
1038COMPARE_OPTIONS_SNAPSHOT_KEY = "_compare_options_snapshot"
1041def trial_options_snapshot_key(key_prefix: str) -> str:
1042 """Session key holding the trial picker's options as it last rendered them."""
1043 return f"_{key_prefix}_trial_options" if key_prefix else "_trial_options"
1046def compare_step_linked() -> bool:
1047 """True when ◀ ▶ should advance scanpath A **and** B (CMP-13).
1049 Both halves matter: the checkbox only exists while compare mode is on, and
1050 Streamlit drops an unrendered widget's key, so a stale ``True`` must not
1051 quietly steer the main picker once the user has left compare mode.
1052 """
1053 return bool(
1054 st.session_state.get("single_compare_toggle")
1055 and st.session_state.get(COMPARE_STEP_LINK_KEY)
1056 )
1059def step_within(options: list[str], state_key: str, delta: int) -> int | None:
1060 """Move ``state_key``'s selection ``delta`` places within ``options``.
1062 Clamped to the ends, and clamped *independently* of any other picker — the
1063 linked step is "advance both", not "keep them aligned": the two pools have
1064 different sizes (B has its own filters, and a cross-dataset B is another
1065 corpus entirely), so their indices carry no shared meaning.
1067 Returns the new index, or ``None`` when there was nothing to step.
1068 """
1069 opts = list(options or [])
1070 if not opts:
1071 return None
1072 try:
1073 pos = opts.index(st.session_state.get(state_key))
1074 except ValueError:
1075 pos = 0
1076 new_pos = max(0, min(pos + delta, len(opts) - 1))
1077 st.session_state[state_key] = opts[new_pos]
1078 return new_pos
1081def at_list_end(options: list[str], state_key: str, delta: int) -> bool:
1082 """True when ``state_key``'s selection cannot move ``delta`` within ``options``.
1084 Used to decide whether a step button is dead. An unknown list answers
1085 **False** so the button stays live: a click that turns out to be a no-op is a
1086 better failure than a button greyed out while the other side could still move.
1087 """
1088 opts = list(options or [])
1089 if not opts:
1090 return False
1091 try:
1092 pos = opts.index(st.session_state.get(state_key))
1093 except ValueError:
1094 return False
1095 return not (0 <= pos + delta < len(opts))
1098def step_linked_compare(delta: int) -> None:
1099 """Advance scanpath **B** by ``delta``, resolved against the list A last saw.
1101 Written as an *identity* rather than an index or a label, because both are
1102 unstable across this step: ``build_comparison_options`` builds B's pool
1103 relative to A (📄 same-text first, then 👤 same-participant), so once A
1104 moves, B's list is re-ordered *and* re-labelled — the same trial can gain or
1105 lose its 📄 marker. Parking the identity in the same
1106 pending slot the ``?compare=`` deep link uses lets the rebuilt picker re-find
1107 the trial the user was actually looking at.
1108 """
1109 from .session_keys import PENDING_COMPARE_STATE_KEY
1111 snapshot = list(st.session_state.get(COMPARE_OPTIONS_SNAPSHOT_KEY) or [])
1112 if not snapshot:
1113 return
1114 labels = [row[0] for row in snapshot]
1115 try:
1116 pos = labels.index(st.session_state.get(COMPARE_TRIAL_KEY))
1117 except ValueError:
1118 pos = 0
1119 _, participant, trial = snapshot[max(0, min(pos + delta, len(snapshot) - 1))]
1120 st.session_state[PENDING_COMPARE_STATE_KEY] = {
1121 "participant_id": participant,
1122 "trial_id": trial,
1123 }
1126#: #374 F34 — ``{"<key prefix>|<dataset>": trial id}``: the trial each
1127#: dataset's picker was last on.
1128_TRIAL_BY_DATASET_KEY = "_trial_by_dataset"
1131def row_tail(column, key_prefix: str, reserve_screen_cell):
1132 """Lay out a selector row's last cell: the screen navigator, then the menus.
1134 ``reserve_screen_cell`` is handed the cell's container to keep a slot for
1135 the screen navigator (filled once the trial is resolved). Returns the
1136 ``railbtn_*`` cluster after it, where the row's ⇅ 🔎 ✏️ go — at the row's
1137 right end, while ◀ ▶ stay beside the slider they step.
1138 """
1139 tail = column.container(
1140 key=f"{key_prefix}_row_tail",
1141 horizontal=True,
1142 vertical_alignment="bottom",
1143 gap="small",
1144 )
1145 reserve_screen_cell(tail)
1146 return tail.container(key=f"railbtn_{key_prefix}_menus", width="content")
1149def _select_trial_none_mode(
1150 combos: pd.DataFrame,
1151 trial_field: str,
1152 text_field: str,
1153 key_prefix: str,
1154 picker_host=None,
1155 *,
1156 words: pd.DataFrame | None = None,
1157 fixations: pd.DataFrame | None = None,
1158 leading_renderer=None,
1159 filter_renderer=None,
1160 trailing_renderer=None,
1161) -> tuple[str | None, str | None, str | None]:
1162 """The trial picker: **dataset + selectbox + scrubbing slider + ◀ ▶ steps + ⇅
1163 sort + 🔎 filters**, all on one row (UX-64). The slider thumb shows ``index/TOTAL · id``
1164 (index first). The pool is narrowed upstream (the "Narrow by" multiselects +
1165 the "More" filters), so this just picks one trial from it and orders it.
1167 Creates its own row of columns, so call it where columns are allowed (the
1168 Scanpath/Corpus body), not nested inside another column. ``picker_host``
1169 (when given) is the container to render into; defaults to the current one.
1170 ``words`` / ``fixations`` (optional) unlock the computed sort keys (UX-10);
1171 without them only column-based orderings are offered.
1173 ``trailing_renderer`` (optional) is handed a slot in one more column at
1174 the row's right end — the multipart screen navigator, which the caller can
1175 only draw once the trial is resolved, so it keeps the slot and fills it
1176 later. The row's ⇅ 🔎 ✏️ then move after it (``row_tail``)."""
1177 host = picker_host if picker_host is not None else st
1178 available_trials = combos.drop_duplicates(subset=[trial_field])
1179 trial_options = sorted(available_trials[trial_field].dropna().astype(str).unique())
1180 if not trial_options:
1181 st.warning(
1182 "No trials match the filters. Clear one, or use ✕ Clear all filters."
1183 )
1184 st.stop()
1186 # Trial id → participant, so the annotation markers (UX-6) can be looked up per
1187 # option (annotations are keyed by (participant, trial)). Mirrors the selection
1188 # below, which resolves the participant the same way (first matching row).
1189 trial_to_pid = dict(
1190 zip(
1191 available_trials[trial_field].astype(str),
1192 available_trials["participant_id"],
1193 strict=True,
1194 )
1195 )
1197 # Populated once the ⇅ popover has rendered (below), and read by the option
1198 # labels — so an active ordering is *visible* in the picker itself rather than
1199 # only inside the popover that set it.
1200 sort_values: dict[str, str] = {}
1202 # UX-187: each id shown part by part, and the help says what the parts are.
1203 id_display, id_part_names = trial_id_layout(
1204 available_trials,
1205 trial_field,
1206 composite_cols=st.session_state.get("_composite_trial_columns") or (),
1207 )
1209 # Read once: a picker lists every trial in the pool, and going through the
1210 # session for each one cost ~0.3 s a rerun at OneStop scale.
1211 store = store_for_prefix()
1213 def _option_label(value: str) -> str:
1214 marks = annotation_markers(trial_to_pid.get(value), value, store=store)
1215 base = id_display.get(value) or _trial_display_label(value)
1216 # #374 F27: the badges follow the trial, so a narrow picker cuts the
1217 # badges rather than the trial.
1218 label = f"{base} {marks}" if marks else base
1219 shown = sort_values.get(value)
1220 return f"{label} · {shown}" if shown else label
1222 # Every label made once, when the order is final (filled below): Streamlit
1223 # formats each option more than once a run, and the slider's twice over.
1224 option_labels: dict[str, str] = {}
1226 def _format_option(value: str) -> str:
1227 label = option_labels.get(value)
1228 return _option_label(value) if label is None else label
1230 n_trials = len(trial_options)
1231 picker_label = "Select trial"
1232 trial_id_key = f"{key_prefix}_trial_id" if key_prefix else None
1233 slider_key = f"{key_prefix}_trial_pos" if key_prefix else "trial_pos"
1235 # The selectbox (`*_trial_id`) is the canonical selection — the deep-link /
1236 # Save-&-restore code seeds it (`_restore_selection`). The slider mirrors it
1237 # and ◀ ▶ step it; all stay in sync via the trial id.
1238 current_label = st.session_state.get(trial_id_key) if trial_id_key else None
1239 # #374 F34: the trial each dataset was last on, so switching away and back
1240 # returns to it rather than to the first trial.
1241 dataset = annotations_dataset(st.session_state)
1242 remembered = st.session_state.setdefault(_TRIAL_BY_DATASET_KEY, {})
1243 last_dataset_key = f"{_TRIAL_BY_DATASET_KEY}_{key_prefix}"
1244 previous = st.session_state.get(last_dataset_key, dataset)
1245 st.session_state[last_dataset_key] = dataset
1246 # A trial carried over from the dataset left behind is not a choice; one a
1247 # link or a restored settings file put there with the switch is.
1248 chosen = st.session_state.pop(f"_{key_prefix}_trial_chosen", None)
1249 carried = (
1250 previous != dataset
1251 and current_label != chosen
1252 and current_label == remembered.get(f"{key_prefix}|{previous}")
1253 )
1254 back_to = remembered.get(f"{key_prefix}|{dataset}")
1255 if (
1256 trial_id_key
1257 and back_to in trial_options
1258 and (carried or current_label not in trial_options)
1259 ):
1260 current_label = back_to
1261 st.session_state[trial_id_key] = current_label
1262 # Seeded rather than chosen: re-seeded to the *sorted* list's first trial
1263 # once the ⇅ order is known (UX-171 — data order's first, not the id's).
1264 seeded = current_label not in trial_options
1265 if seeded:
1266 current_label = trial_options[0]
1267 if trial_id_key:
1268 st.session_state[trial_id_key] = current_label
1270 idx_of = {opt: i for i, opt in enumerate(trial_options)}
1272 if n_trials > 1:
1273 # Mirror the slider to the current selection BEFORE it renders, so picking
1274 # a trial in the dropdown moves the slider too; the drag callback writes
1275 # the selectbox key, reconciling next run.
1276 st.session_state[slider_key] = current_label
1278 def _on_trial_slider() -> None:
1279 if trial_id_key:
1280 st.session_state[trial_id_key] = st.session_state[slider_key]
1282 def _step_trial(delta: int) -> None:
1283 # ◀ / ▶ : move the canonical selection one trial earlier/later.
1284 if not trial_id_key:
1285 return
1286 step_within(trial_options, trial_id_key, delta)
1287 # CMP-13: while the link is armed, the same ±1 also moves scanpath B.
1288 if compare_step_linked():
1289 step_linked_compare(delta)
1291 def _slider_label(value: str) -> str:
1292 # index/TOTAL first, then the id — the slider doubles as the counter,
1293 # so a separate "Trial X / N" caption is redundant.
1294 return f"{idx_of.get(value, 0) + 1}/{n_trials} · {_format_option(value)}"
1296 # UX-64 — ONE row for everything: [dataset] [trial] [slider] [◀ ▶ ⇅ 🔎].
1297 # The Narrow-by row above it is gone; its filters live in the 🔎 popover
1298 # at the end of this row. The dataset picker keeps its width on purpose
1299 # (you may be comparing two datasets, and the label is what tells them
1300 # apart) — the slider gives up the room instead, and the filter is an
1301 # icon, which is what makes six controls fit.
1302 #
1303 # The three triggers share ONE trailing column as a `railbtn_*` cluster
1304 # (UX-27), which styles.py packs right at a uniform 3px spacing. A column
1305 # each put a full gutter between them, so a prev/next *pair* didn't read
1306 # as a pair.
1307 lead_col, sel_col, slider_col, trail_col, *extra = host.columns(
1308 SELECTOR_ROW_GRID + ([SELECTOR_SCREEN_TRACK] if trailing_renderer else []),
1309 vertical_alignment="bottom",
1310 )
1311 if leading_renderer is not None:
1312 leading_renderer(lead_col)
1313 menus = (
1314 row_tail(extra[0], key_prefix, trailing_renderer)
1315 if trailing_renderer is not None
1316 else None
1317 )
1318 trail = trail_col.container(
1319 key=f"railbtn_{key_prefix}_trail{'_steps' if menus else ''}"
1320 )
1321 # Created in display order (◀ ▶ then ⇅) but filled out of order: the sort
1322 # popover has to render first, because the order it returns is what the
1323 # selectbox, the slider and the ◀ ▶ steps all walk. With a screen cell,
1324 # ⇅ 🔎 ✏️ close the row after it, and ◀ ▶ stay by the slider.
1325 step_col = trail.container(key=f"railbtn_{key_prefix}_step")
1326 cluster = trail if menus is None else menus
1327 sort_col = cluster.container(key=f"railbtn_{key_prefix}_sort")
1328 filter_col = cluster.container(key=f"railbtn_{key_prefix}_filter")
1329 # Filled by the caller, which owns the filter widgets — but created here,
1330 # in display order, so 🔎 lands after ⇅ in the cluster (UX-64).
1331 if filter_renderer is not None:
1332 filter_renderer(filter_col)
1333 # UX-10: order the pool *before* the widgets read `trial_options`, so the
1334 # selectbox, the slider and the ◀ ▶ steps all walk the same order. The
1335 # canonical selection is a trial *id*, so re-sorting never changes which
1336 # trial is selected — only where it sits in the list.
1337 # UX-27: each of the three step/sort triggers goes in a `railbtn_*`
1338 # container so styles.py can give the whole cluster above the plot one
1339 # button shape — these were square (each `width="stretch"` inside its own
1340 # narrow column) beside pill-shaped labelled buttons on the rows above
1341 # and below.
1342 sort_key, sort_desc, sort_choice = _render_trial_sort_popover(
1343 sort_col,
1344 combos,
1345 trial_field,
1346 key_prefix,
1347 words=words,
1348 fixations=fixations,
1349 )
1350 if sort_key is not None:
1351 trial_options = sort_trial_options(
1352 trial_options, sort_key, descending=sort_desc
1353 )
1354 idx_of = {opt: i for i, opt in enumerate(trial_options)}
1355 # UX-171: data order is the default and its values are bare ranks,
1356 # so it carries no per-option value and names itself only reversed.
1357 if sort_choice != TRIAL_SORT_DATA_ORDER:
1358 lookup = sort_key.to_dict()
1359 sort_values.update(
1360 {opt: format_sort_value(lookup.get(opt)) for opt in trial_options}
1361 )
1362 if sort_choice != TRIAL_SORT_DATA_ORDER or sort_desc:
1363 picker_label = (
1364 f"Select trial · by {sort_choice} {'↓' if sort_desc else '↑'}"
1365 )
1366 if seeded:
1367 current_label = trial_options[0]
1368 if trial_id_key:
1369 st.session_state[trial_id_key] = current_label
1370 st.session_state[slider_key] = current_label
1371 current_idx = trial_options.index(current_label)
1372 else:
1373 # A one-trial pool has no slider (`st.select_slider` throws on a single
1374 # option — BUG-23) and nothing to step through, but it still needs the
1375 # dataset picker and the filters: a pool of one is *usually the result of
1376 # a filter*, so this is exactly when the user reaches for them. UX-64.
1377 lead_col, sel_col, trail_col, *extra = host.columns(
1378 SELECTOR_ROW_TRIO + ([SELECTOR_SCREEN_TRACK] if trailing_renderer else []),
1379 vertical_alignment="bottom",
1380 )
1381 if leading_renderer is not None:
1382 leading_renderer(lead_col)
1383 menus = (
1384 row_tail(extra[0], key_prefix, trailing_renderer)
1385 if trailing_renderer is not None
1386 else None
1387 )
1388 if filter_renderer is not None:
1389 filter_renderer(
1390 (trail_col if menus is None else menus).container(
1391 key=f"railbtn_{key_prefix}_filter_solo"
1392 )
1393 )
1395 # CMP-13: publish the list as rendered (post-sort), so the *Compare To*
1396 # picker's linked ◀ ▶ can step this picker without rebuilding its ordering.
1397 if trial_id_key:
1398 st.session_state[trial_options_snapshot_key(key_prefix)] = list(trial_options)
1399 # #374 F10: the browser identifies the picked option by its *label*, and
1400 # a label carries the trial's ★ 🏷️ 📝 marks. Written only when it moved,
1401 # the browser kept the label from that run; a tag added later renamed the
1402 # option, the old label matched nothing, and the view fell back to trial
1403 # 1. Written every run (as the slider is), the browser always holds the
1404 # label it was last shown, which the next run can still read back.
1405 st.session_state[trial_id_key] = current_label
1406 remembered[f"{key_prefix}|{dataset}"] = current_label
1408 option_labels.update({opt: _option_label(opt) for opt in trial_options})
1410 # The label is shown so its help "?" icon (the type-to-search hint) is visible
1411 # — a collapsed label hides it.
1412 selected_trial_label = sel_col.selectbox(
1413 picker_label,
1414 options=trial_options,
1415 key=trial_id_key,
1416 # BUG-80: the picker renders only on the Scanpath view, and Streamlit
1417 # drops an unrendered widget's key at the end of the run — so a trip to
1418 # Corpus Analysis or 🗂️ Data came back on trial 1 (and, in Compare,
1419 # left A and B on different texts). A value the pool no longer holds is
1420 # still reset above, before this renders.
1421 persist_state="session",
1422 format_func=_format_option,
1423 help=" ".join(
1424 filter(
1425 None,
1426 (
1427 trial_id_help(id_part_names),
1428 "Click this dropdown, then type to narrow the list. "
1429 "★ favorite · 🏷️ tagged · 📝 has notes. When a sort key is "
1430 "active, each option ends with that trial's value for it.",
1431 ),
1432 )
1433 ),
1434 )
1435 if trial_id_key:
1436 # The menu opens as wide as its longest trial id (+ marks / sort value).
1437 widen_menu(trial_id_key, option_labels.values())
1439 if n_trials > 1:
1440 slider_labels = {opt: _slider_label(opt) for opt in trial_options}
1441 with slider_col:
1442 st.select_slider(
1443 "Trial",
1444 options=trial_options,
1445 key=slider_key,
1446 persist_state="session",
1447 on_change=_on_trial_slider,
1448 help=f"Scrub through the {n_trials} trials (index/total · id, "
1449 "plus the sort value when one is active); the dropdown jumps to "
1450 "a specific id.",
1451 label_visibility="collapsed",
1452 format_func=lambda value: (
1453 slider_labels.get(value) or _slider_label(value)
1454 ),
1455 )
1456 # Both step buttons in the keyed container reserved above, which styles.py
1457 # lays out as a flex ROW (a Streamlit vertical block stacks its children
1458 # by default).
1459 steps = step_col
1460 # CMP-13: linked, a button stays live until BOTH sides have run out — a
1461 # side at the end of its own list just stays put while the other keeps
1462 # stepping. `at_list_end` answers False for a list it can't see, so the
1463 # worst case is a click that moves only one scanpath.
1464 linked = compare_step_linked()
1465 compare_snapshot = (
1466 [
1467 row[0]
1468 for row in (st.session_state.get(COMPARE_OPTIONS_SNAPSHOT_KEY) or [])
1469 ]
1470 if linked
1471 else []
1472 )
1473 step_help = " Linked: also steps the compared trial." if linked else ""
1474 # UX-200: `spoken` names the glyph buttons for screen readers.
1475 steps.button(
1476 f"◀ {spoken('Previous trial')}",
1477 key=f"{key_prefix}_prev_trial" if key_prefix else "prev_trial",
1478 wrap=True,
1479 on_click=_step_trial,
1480 args=(-1,),
1481 disabled=current_idx == 0
1482 and (not linked or at_list_end(compare_snapshot, COMPARE_TRIAL_KEY, -1)),
1483 help="Previous trial." + step_help,
1484 )
1485 steps.button(
1486 f"▶ {spoken('Next trial')}",
1487 key=f"{key_prefix}_next_trial" if key_prefix else "next_trial",
1488 wrap=True,
1489 on_click=_step_trial,
1490 args=(1,),
1491 disabled=current_idx == n_trials - 1
1492 and (not linked or at_list_end(compare_snapshot, COMPARE_TRIAL_KEY, 1)),
1493 help="Next trial." + step_help,
1494 )
1496 if not selected_trial_label:
1497 return None, None, None
1499 chosen = available_trials[
1500 available_trials[trial_field].astype(str) == selected_trial_label
1501 ].iloc[0]
1502 selected_text = str(chosen[text_field]) if text_field in chosen.index else None
1503 return chosen["participant_id"], chosen["trial_id"], selected_text
1506def select_trial(
1507 combos: pd.DataFrame,
1508 key_prefix: str = "",
1509 picker_host=None,
1510 *,
1511 words: pd.DataFrame | None = None,
1512 fixations: pd.DataFrame | None = None,
1513 leading_renderer=None,
1514 filter_renderer=None,
1515 trailing_renderer=None,
1516) -> tuple[str | None, str | None, str, str | None]:
1517 """Pick a specific trial from the (already-narrowed) pool.
1519 There are no Browse-by modes anymore, and no per-mapping variants either: the
1520 pool is narrowed by the inline **Narrow by** Text / Participant multiselects +
1521 the **More** filters (``controls.render_narrow_by`` /
1522 ``render_trial_filters``), and this always picks one trial via the same
1523 selectbox + slider + ◀ ▶ arrows — whatever the Trial ID mapping looks like.
1525 BUG-23: a composite trial id (built from several mapped columns) used to get a
1526 picker of its own — first one selector per mapped component, then a Participant
1527 → Text cascade. Both made the *shape of the mapping* visible in the UI, and
1528 neither offered the slider or the step buttons, so stepping through trials
1529 worked on some datasets and not others. Participant and Text are what **Narrow
1530 by** is for; the composite flag now only tells the chip strip to spell the
1531 joined id out (``tabs._render_trial_condition_chips``).
1533 ``picker_host`` is the container to render into (defaults to the current one);
1534 the picker builds its own row of columns, so call it where columns are allowed.
1536 ``words`` / ``fixations`` (optional) are the frames ``combos`` was built from;
1537 passing them unlocks the UX-10 ⇅ sort popover's computed keys (fixation count,
1538 reading time). Without them the sort still offers the column-based orderings.
1540 Returns:
1541 Tuple of (participant_id, trial_id, selection_mode, selected_text).
1542 ``selection_mode`` is always ``"Trial"`` (kept for the comparison-options
1543 builder, which still supports the other modes when called directly).
1544 """
1545 if combos.empty:
1546 st.warning(
1547 "No trials match the filters. Clear one, or use ✕ Clear all filters."
1548 )
1549 st.stop()
1551 trial_field = (
1552 "unique_trial_id" if "unique_trial_id" in combos.columns else "trial_id"
1553 )
1554 text_field = "unique_text_id" if "unique_text_id" in combos.columns else "text_id"
1556 participant, trial, text = _select_trial_none_mode(
1557 combos,
1558 trial_field,
1559 text_field,
1560 key_prefix,
1561 picker_host=picker_host,
1562 # UX-64: the row's lead cell (the dataset picker) and its 🔎 filter
1563 # popover are filled by the caller, which owns those widgets.
1564 leading_renderer=leading_renderer,
1565 filter_renderer=filter_renderer,
1566 trailing_renderer=trailing_renderer,
1567 words=words,
1568 fixations=fixations,
1569 )
1571 return participant, trial, "Trial", text
1574# -----------------------------------------------------------------------------
1575# Statistics and metadata
1576# -----------------------------------------------------------------------------
1579def compute_trial_stats(
1580 trial_words: pd.DataFrame, trial_fixations: pd.DataFrame
1581) -> dict[str, float]:
1582 """Compute summary statistics for a single trial."""
1583 total_time = None
1584 if "trial_dwell_time_ms" in trial_words.columns:
1585 dwell_values = (
1586 pd.to_numeric(trial_words["trial_dwell_time_ms"], errors="coerce")
1587 .dropna()
1588 .unique()
1589 )
1590 if len(dwell_values):
1591 total_time = float(dwell_values[0])
1592 if total_time is None:
1593 total_time = (
1594 float(trial_fixations["duration_ms"].sum())
1595 if not trial_fixations.empty
1596 else 0.0
1597 )
1598 return dict(
1599 total_reading_time_ms=total_time,
1600 total_reading_time_s=total_time / 1000.0,
1601 word_count=len(trial_words),
1602 fixation_count=len(trial_fixations),
1603 )
1606def safe_summary(series: pd.Series) -> dict:
1607 """Compute summary statistics for a series, handling empty data."""
1608 if series.empty:
1609 nan_val = float("nan")
1610 return dict(mean=nan_val, std=nan_val, min=nan_val, max=nan_val, median=nan_val)
1611 return dict(
1612 mean=float(series.mean()),
1613 std=float(series.std(ddof=0)),
1614 min=float(series.min()),
1615 max=float(series.max()),
1616 median=float(series.median()),
1617 )
1620# -----------------------------------------------------------------------------
1621# Comparison helpers
1622# -----------------------------------------------------------------------------
1625# Markers shown beside comparison-trial options.
1626SAME_TEXT_MARKER = (
1627 "📄" # same stimulus text as the primary trial (★ reserved for favorites, UX-6)
1628)
1629SAME_PARTICIPANT_MARKER = "👤" # same participant as the primary trial
1632def _allocate_label(label: str, qualified: str, used_labels: set[str]) -> str:
1633 """``label`` if unused, else ``qualified``, else ``qualified (n)`` for the
1634 first free ``n`` — always a label no earlier option holds (round 11: the
1635 qualified form itself could already be taken, by a trial id that reads
1636 like it, and the later option then overwrote the earlier one's identity)."""
1637 candidate = label if label not in used_labels else qualified
1638 n = 2
1639 while candidate in used_labels:
1640 candidate = f"{qualified} ({n})"
1641 n += 1
1642 used_labels.add(candidate)
1643 return candidate
1646def _compare_option_label(
1647 participant_id: str,
1648 trial_id: str,
1649 markers: str,
1650 used_labels: set[str],
1651) -> str:
1652 """Selectbox label for a comparison option: ``"<markers> <trial_id>"``.
1654 De-duplicates on ``trial_id`` (two participants can share one) by appending
1655 the participant in brackets — and a counter when even that is taken — so the
1656 label stays a unique selectbox option / dict key in
1657 ``tabs._render_compare_selector``."""
1658 trial_str = str(trial_id) if trial_id is not None else ""
1659 prefix = f"{markers} " if markers else ""
1660 return _allocate_label(
1661 f"{prefix}{trial_str}", f"{prefix}{trial_str} [{participant_id}]", used_labels
1662 )
1665def friendly_trial_label(
1666 participant_id: str,
1667 trial_id: str,
1668 text_id: str | None,
1669 existing_labels: set[str],
1670 prefix: str = "",
1671) -> str:
1672 """Create a short, de-duplicated label for comparison dropdowns/legends."""
1673 trial_str = str(trial_id) if trial_id is not None else ""
1674 text_str = str(text_id) if text_id is not None else ""
1675 text_str = text_str.strip()
1676 trial_contains_text = text_str and text_str.lower() in trial_str.lower()
1678 if text_str:
1679 base = f"{text_str} · {participant_id}"
1680 if not trial_contains_text:
1681 base = f"{base} (trial {trial_str})" if trial_str else base
1682 elif trial_str != text_str:
1683 # Surface any trial_id suffix beyond the text id (e.g. a
1684 # repeat-reading "_r2" tag added during normalization). Without
1685 # this the primary and compare titles look identical when a
1686 # participant re-read the same text.
1687 extra = trial_str
1688 if extra.lower().startswith(text_str.lower()):
1689 extra = extra[len(text_str) :].lstrip("_- ")
1690 if extra:
1691 base = f"{text_str} ({extra}) · {participant_id}"
1692 else:
1693 base = f"{trial_str} · {participant_id}" if trial_str else participant_id
1695 return _allocate_label(
1696 f"{prefix}{base}", f"{prefix}{base} [{trial_str or 'trial'}]", existing_labels
1697 )
1700def build_comparison_options(
1701 combos: pd.DataFrame,
1702 selection_mode: str,
1703 primary_participant: str,
1704 primary_trial: str,
1705 primary_text: str | None,
1706 *,
1707 cross_dataset: bool = False,
1708 include_primary: bool = True,
1709) -> list[tuple[str, str, str, str]]:
1710 """Build a prioritized list of comparison-trial options.
1712 Returns ``(participant_id, trial_id, label, markers)`` tuples, where
1713 ``markers`` leads with the relation icons ``"📄"`` (same text) / ``"👤"`` (same
1714 participant) and then the trial's annotation markers ``★`` (favorite) / ``🏷️``
1715 (tagged) / ``📝`` (noted) when present (UX-6), and ``label`` is
1716 ``"<markers> <trial_id>"``. Ordered: same-text (📄) first, then same-participant
1717 (👤), then the rest. A trial that is BOTH same-text and same-participant sorts
1718 with the 📄 group (text-matches lead) and shows both markers.
1720 ``cross_dataset`` (CMP-8 §5.1) says ``combos`` describes a *different*
1721 dataset, and degrades the three id-based signals that would otherwise lie:
1722 two corpora do not share readers, so 👤 never fires; a foreign trial's ★ /
1723 🏷️ / 📝 are read from *its* dataset's annotations, never the active
1724 dataset's, whose matching-looking ids name other trials (DATA-48 — they
1725 were dropped altogether before annotations were per dataset); and the
1726 primary trial is not in this pool, so a
1727 coincidentally identical ``(participant, trial)`` is a real candidate rather
1728 than the trial being compared. 📄 survives — a text id that matches across
1729 corpora is exactly the pairing this feature exists for.
1731 ``include_primary`` (CMP-22) keeps the selected trial itself in the pool, so
1732 B's picker lists every trial A's does and the two position readouts agree.
1733 It is only a *candidate*: the picker defaults B to the first trial that is
1734 not A. Pass ``False`` to ask "is there anything else to compare with?" —
1735 the question the Compare gate asks.
1736 """
1737 text_field = "unique_text_id" if "unique_text_id" in combos.columns else "text_id"
1738 uniq = combos.drop_duplicates(subset=["participant_id", "trial_id"])
1740 foreign_store = store_for_prefix("cmp") if cross_dataset else None
1741 rows: list[dict] = []
1742 for row in uniq.itertuples():
1743 if (
1744 not include_primary
1745 and not cross_dataset
1746 and (row.participant_id, row.trial_id)
1747 == (primary_participant, primary_trial)
1748 ):
1749 continue
1750 text_id = getattr(row, text_field, "")
1751 same_text = bool(primary_text and str(text_id) == str(primary_text))
1752 same_participant = not cross_dataset and bool(
1753 str(row.participant_id) == str(primary_participant)
1754 )
1755 markers = (
1756 (SAME_TEXT_MARKER if same_text else "")
1757 + (SAME_PARTICIPANT_MARKER if same_participant else "")
1758 + annotation_markers(row.participant_id, row.trial_id, store=foreign_store)
1759 )
1760 rows.append(
1761 {
1762 "participant_id": row.participant_id,
1763 "trial_id": row.trial_id,
1764 "same_text": same_text,
1765 "same_participant": same_participant,
1766 "markers": markers,
1767 }
1768 )
1770 # ★ group first, then 👤 group, then the rest. same_text is the primary sort
1771 # key so a both-★-👤 trial leads the ★ group. Stable sort preserves combos
1772 # order within a group.
1773 rows.sort(
1774 key=lambda r: (0 if r["same_text"] else 1, 0 if r["same_participant"] else 1)
1775 )
1777 used_labels: set[str] = set()
1778 options: list[tuple[str, str, str, str]] = []
1779 for r in rows:
1780 label = _compare_option_label(
1781 r["participant_id"], r["trial_id"], r["markers"], used_labels
1782 )
1783 options.append((r["participant_id"], r["trial_id"], label, r["markers"]))
1784 return options
1787# -----------------------------------------------------------------------------
1788# Cross-dataset comparison frames (CMP-8 · promoted from tabs.py for CMP-9)
1789# -----------------------------------------------------------------------------
1790# These live here, not in tabs.py, because three surfaces now build a
1791# cross-dataset comparison — the app, `api.compare_scanpaths` and
1792# `cli.render --compare-*` — and the namespacing rule below is the one piece of
1793# it that must not be re-derived per surface. `tests/test_compare_cross_dataset.py`
1794# exists because getting it wrong renders a *plausible-looking wrong figure*
1795# rather than an error.
1797#: Separator between a dataset name and a participant id in a qualified id.
1798COMPARE_DATASET_SEP = " · "
1801def qualify_for_compare(frame: pd.DataFrame, dataset: str) -> pd.DataFrame:
1802 """A copy of ``frame`` whose ``participant_id`` is namespaced by ``dataset``.
1804 Two corpora can hold the same ``(participant_id, trial_id)``, and
1805 `plots.make_comparison_figure` slices its frame by exactly that pair — so an
1806 unqualified merge would silently render *the wrong scanpath*, or two.
1808 Only ever applied to the single-trial frames that feed the comparison
1809 builder. Nothing the annotations, the export slug, the deep link or Corpus
1810 Analysis reads goes through here: those key on the real ids, and must.
1811 """
1812 if frame.empty:
1813 return frame
1814 out = frame.copy()
1815 out["dataset"] = dataset
1816 out["participant_id"] = (
1817 dataset + COMPARE_DATASET_SEP + out["participant_id"].astype(str)
1818 )
1819 return out
1822def separate_self_compare(frame: pd.DataFrame, participant: str) -> pd.DataFrame:
1823 """A copy of B's single-trial ``frame`` renamed apart from A's (CMP-22).
1825 B may now be A's own trial. `plots.make_comparison_figure` slices its merged
1826 frame by ``(participant_id, trial_id)``, so two copies of one trial would hand
1827 *each* side both copies (and a duplicated index the word-line clustering
1828 rejects). Giving B's copy `self_compare_participant`'s id keeps the halves
1829 apart — the same trick `qualify_for_compare` plays across corpora, and just
1830 as figure-only: labels, lookups, exports and links keep the real id.
1831 """
1832 if frame.empty:
1833 return frame
1834 return frame.assign(participant_id=self_compare_participant(participant))
1837def self_compare_participant(participant: str) -> str:
1838 """The id `separate_self_compare` gives B's copy of ``participant``."""
1839 return f"{participant}{COMPARE_DATASET_SEP}B"
1842def qualified_participant(dataset: str, participant: str) -> str:
1843 """The id `qualify_for_compare` gives ``participant`` inside ``dataset``."""
1844 return f"{dataset}{COMPARE_DATASET_SEP}{participant}"
1847def unqualify_for_export(frame: pd.DataFrame, participant: str) -> pd.DataFrame:
1848 """Undo `qualify_for_compare`'s rename, restoring the corpus' own id.
1850 The namespace exists so `make_comparison_figure` can slice two colliding
1851 ``(participant, trial)`` pairs apart. An exported table must carry the id the
1852 corpus actually uses, or it won't join back to anything (CMP-8 §6). The
1853 stamped ``dataset`` column is kept — that is what disambiguates the rows.
1854 """
1855 if frame is None or frame.empty or "dataset" not in frame.columns:
1856 return frame
1857 out = frame.copy()
1858 out["participant_id"] = str(participant)
1859 return out
1862def align_compare_columns(
1863 a: pd.DataFrame, b: pd.DataFrame
1864) -> tuple[pd.DataFrame, pd.DataFrame, frozenset]:
1865 """Reindex two frames onto their column **union**, and report shared numerics.
1867 A bare ``pd.concat`` of frames with disjoint columns warns and churns dtypes
1868 (int columns become float once the other frame's rows fill in as NaN), which
1869 matters here because two corpora rarely ship the same measure set. Aligning
1870 first keeps the concat quiet and the dtypes stable.
1872 The third element is the columns both frames carry *as the same kind* —
1873 numeric in both, or categorical in both: what a cross-dataset figure may
1874 legitimately colour by (CMP-8 §5.4; categorical since Compare colours by a
1875 category too). A column present in only one corpus would colour one panel
1876 and blank the other, and one numeric on one side only would be a scale on
1877 one panel and a palette on the other.
1878 """
1879 union = list(dict.fromkeys([*a.columns, *b.columns]))
1880 a_aligned = a.reindex(columns=union) if list(a.columns) != union else a
1881 b_aligned = b.reindex(columns=union) if list(b.columns) != union else b
1882 shared = frozenset(
1883 col
1884 for col in set(a.columns) & set(b.columns)
1885 if pd.api.types.is_numeric_dtype(a[col])
1886 == pd.api.types.is_numeric_dtype(b[col])
1887 )
1888 return a_aligned, b_aligned, shared