Coverage for scanpath_studio/utils.py: 96%

649 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Utility functions for trial selection, statistics, and labelling.""" 

2 

3from __future__ import annotations 

4 

5import re 

6from collections.abc import Callable, Iterable 

7 

8import numpy as np 

9import pandas as pd 

10import streamlit as st 

11 

12from . import progress 

13from .annotations import current_dataset as annotations_dataset 

14from .annotations import get_entry, store_for_prefix 

15from .column_names import COMPUTED_SUFFIX, active_all 

16from .constants import ( 

17 SELECTOR_ROW_GRID, 

18 SELECTOR_ROW_TRIO, 

19 SELECTOR_SCREEN_TRACK, 

20 spoken, 

21) 

22from .data import frame_fingerprint, stable_id 

23from .fields import labeled 

24from .styles import widen_menu 

25 

26# Annotation markers shown beside a trial in the pickers (UX-6). Independent of 

27# the same-text/same-participant markers (UX-4) and of each other — a trial can 

28# carry any combination, so they compose. ★ favorite · 🏷️ tagged · 📝 noted. 

29FAVORITE_MARKER = "★" 

30TAGGED_MARKER = "🏷️" 

31NOTE_MARKER = "📝" 

32 

33 

34def annotation_markers(participant_id, trial_id, *, store=None) -> str: 

35 """Composable annotation markers (★ favorite · 🏷️ tagged · 📝 noted) for a 

36 trial, or ``""`` when it carries no annotations. Reads the session store — 

37 the open dataset's — or ``store``, another dataset's (DATA-48).""" 

38 if participant_id is None or trial_id is None: 

39 return "" 

40 entry = ( 

41 get_entry(str(participant_id), str(trial_id)) 

42 if store is None 

43 else store.get((str(participant_id), str(trial_id))) or {} 

44 ) 

45 marks = "" 

46 if entry.get("star"): 

47 marks += FAVORITE_MARKER 

48 if entry.get("tags"): 

49 marks += TAGGED_MARKER 

50 if str(entry.get("note") or "").strip(): 

51 marks += NOTE_MARKER 

52 return marks 

53 

54 

55# ----------------------------------------------------------------------------- 

56# Trial combo building 

57# ----------------------------------------------------------------------------- 

58 

59#: The identity columns `build_combo_options` reads off a frame (besides any 

60#: composite-trial components) — all `combo_source` has to carry. 

61_COMBO_ID_COLUMNS = ( 

62 "participant_id", 

63 "trial_id", 

64 "unique_trial_id", 

65 "unique_text_id", 

66 "text_id", 

67 "unique_paragraph_id", 

68 "paragraph_id", 

69 "TRIAL_INDEX", 

70 "trial_index", 

71) 

72 

73 

74def combo_source( 

75 fixations: pd.DataFrame, 

76 words: pd.DataFrame, 

77 raw_gaze: pd.DataFrame | None = None, 

78) -> pd.DataFrame: 

79 """The frame the trial picker's combos are built from. 

80 

81 Fixations when there are any, else words (a words-only dataset), else raw 

82 gaze (a raw-gaze-only one) — and, since VIZ-45, **plus the trials only the 

83 raw gaze has**. A trial recorded as samples alone is a trial: in a dataset 

84 whose fixations cover other trials, or whose words table covers other 

85 texts, it used to be unpickable because the picker listed the first 

86 non-empty table and nothing else. 

87 

88 Returns the chosen frame itself whenever the raw gaze adds no trial (the 

89 common case, and every dataset without raw gaze), so `build_combo_options` 

90 keys its cache on the same object as before. Otherwise it returns a small 

91 frame of identity rows — the chosen frame's, in their order, then the 

92 raw-gaze-only trials' — which is all `build_combo_options` reads. 

93 """ 

94 for primary in (fixations, words): 

95 if primary is not None and not primary.empty: 

96 break 

97 else: 

98 return raw_gaze if raw_gaze is not None else pd.DataFrame() 

99 if raw_gaze is None or raw_gaze.empty: 

100 return primary 

101 composite_cols = tuple(st.session_state.get("_composite_trial_columns") or []) 

102 combined = _combo_source_with_raw_gaze( 

103 primary, 

104 raw_gaze, 

105 composite_cols, 

106 cache_key=(frame_fingerprint(primary), frame_fingerprint(raw_gaze)), 

107 ) 

108 return primary if combined is None else combined 

109 

110 

111@st.cache_data(show_spinner=False, max_entries=16) 

112def _combo_source_with_raw_gaze( 

113 _primary: pd.DataFrame, 

114 _raw_gaze: pd.DataFrame, 

115 composite_cols: tuple[str, ...], 

116 cache_key, 

117) -> pd.DataFrame | None: 

118 """`combo_source`'s identity rows, or ``None`` when raw gaze adds no trial.""" 

119 progress.report() 

120 from .data import trial_keys 

121 

122 # One deduplication per table, on the identity columns only; every key 

123 # set below comes off those small frames rather than another pass over 

124 # every sample (PERF: three full scans at 5M samples was ~0.8 s a miss). 

125 wanted = [*_COMBO_ID_COLUMNS, *composite_cols] 

126 primary_cols = [c for c in dict.fromkeys(wanted) if c in _primary.columns] 

127 rows = _primary[primary_cols].drop_duplicates() 

128 raw_cols = [c for c in dict.fromkeys(wanted) if c in _raw_gaze.columns] 

129 raw_rows = _raw_gaze[raw_cols].drop_duplicates() 

130 extra_keys = trial_keys(raw_rows) - trial_keys(rows) 

131 if not extra_keys: 

132 return None 

133 index = pd.MultiIndex.from_arrays( 

134 [raw_rows["participant_id"].astype(str), raw_rows["trial_id"].astype(str)] 

135 ) 

136 raw_rows = raw_rows[index.isin(extra_keys)].copy() 

137 # The picker keys on the primary frame's trial and text columns; a raw-gaze 

138 # row that lacks one takes its own trial id / text id, which is what those 

139 # columns mean for a normalized frame (`data.normalize_raw_gaze`). 

140 if "unique_trial_id" in rows.columns and "unique_trial_id" not in raw_rows: 

141 raw_rows["unique_trial_id"] = raw_rows["trial_id"] 

142 for text_col in ("unique_text_id", "text_id", "unique_paragraph_id"): 

143 if text_col in rows.columns and text_col not in raw_rows.columns: 

144 raw_rows[text_col] = raw_rows.get("text_id", raw_rows["trial_id"]) 

145 return pd.concat([rows, raw_rows], ignore_index=True) 

146 

147 

148def build_combo_options( 

149 fixations: pd.DataFrame, 

150) -> tuple[pd.DataFrame, list[str], dict[str, tuple[str, str]]]: 

151 """Build participant/trial/text combinations for selection UI. 

152 

153 Returns: 

154 Tuple of (combos DataFrame, label list, label-to-combo mapping). 

155 

156 Cached on a cheap fingerprint of the frame + the composite-trial columns, so 

157 the full-frame ``drop_duplicates`` + label build don't re-run on every rerun 

158 (e.g. selecting a different trial). The session-state read happens here, in 

159 the un-cached wrapper, and is threaded into the cached core as an argument. 

160 """ 

161 composite_cols = tuple(st.session_state.get("_composite_trial_columns") or []) 

162 return build_combo_options_for(fixations, composite_cols) 

163 

164 

165def build_combo_options_for( 

166 fixations: pd.DataFrame, 

167 composite_cols: tuple[str, ...] = (), 

168) -> tuple[pd.DataFrame, list[str], dict[str, tuple[str, str]]]: 

169 """`build_combo_options` for a frame that is **not** the active dataset. 

170 

171 CMP-8 §2: a comparison source has its own composite-trial columns, so the 

172 session-state read in `build_combo_options` would answer for the wrong 

173 dataset. Everything else — the cache key, the cached core — is shared. 

174 """ 

175 composite_cols = tuple(composite_cols or ()) 

176 return _build_combo_options_cached( 

177 fixations, 

178 composite_cols, 

179 cache_key=(frame_fingerprint(fixations), composite_cols), 

180 ) 

181 

182 

183# UX-166: the dataset card lists this step. 

184@st.cache_data(show_spinner=False) 

185def _build_combo_options_cached( 

186 _fixations: pd.DataFrame, 

187 composite_cols: tuple[str, ...], 

188 cache_key, 

189) -> tuple[pd.DataFrame, list[str], dict[str, tuple[str, str]]]: 

190 progress.report() # a miss: real work, so a gated card over it may show 

191 fixations = _fixations 

192 trial_col = ( 

193 "unique_trial_id" if "unique_trial_id" in fixations.columns else "trial_id" 

194 ) 

195 # The text/passage column is optional. Normalized frames carry a text_id (it 

196 # falls back to trial_id when no text is mapped), but a frame may arrive with 

197 # only the source name (e.g. unique_paragraph_id — the pre-rename text id, and 

198 # which can also be a composite-trial component). Detect it via the same 

199 # priority list normalization uses, and *copy* it to text_id rather than 

200 # renaming so a shared composite component column survives for the picker. 

201 text_col = next( 

202 ( 

203 c 

204 for c in ( 

205 "unique_text_id", 

206 "text_id", 

207 "unique_paragraph_id", 

208 "paragraph_id", 

209 ) 

210 if c in fixations.columns 

211 ), 

212 None, 

213 ) 

214 combo_cols = ["participant_id", trial_col] 

215 if text_col is not None and text_col not in combo_cols: 

216 combo_cols.append(text_col) 

217 for col in ["unique_trial_id", "unique_text_id", "TRIAL_INDEX", "trial_index"]: 

218 if col in fixations.columns and col not in combo_cols: 

219 combo_cols.append(col) 

220 # Carry the composite trial id's component columns through, so the trial 

221 # picker can detect a composite id and cascade on identity (see select_trial). 

222 for col in composite_cols: 

223 if col in fixations.columns and col not in combo_cols: 

224 combo_cols.append(col) 

225 

226 # UX-24: preserve each trial's first appearance before the historical 

227 # participant/id sort. The visible pool may still default to Trial ID, but 

228 # the ⇅ menu can now reconstruct source-file order exactly. 

229 combos = fixations[combo_cols].drop_duplicates().copy() 

230 combos["_data_order"] = np.arange(len(combos), dtype=int) 

231 combos = combos.rename(columns={trial_col: "trial_id"}) 

232 if "text_id" not in combos.columns: 

233 combos["text_id"] = ( 

234 combos[text_col] if text_col is not None else combos["trial_id"] 

235 ) 

236 if trial_col == "unique_trial_id" and "unique_trial_id" not in combos.columns: 

237 combos["unique_trial_id"] = combos["trial_id"] 

238 if text_col == "unique_text_id" and "unique_text_id" not in combos.columns: 

239 combos["unique_text_id"] = combos["text_id"] 

240 sort_cols = ["participant_id"] 

241 if "TRIAL_INDEX" in combos.columns: 

242 sort_cols.append("TRIAL_INDEX") 

243 elif "trial_index" in combos.columns: 

244 sort_cols.append("trial_index") 

245 sort_cols.append("trial_id") 

246 combos = combos.sort_values(sort_cols) 

247 

248 combo_labels = [ 

249 f"{row.participant_id} / {row.trial_id} · {row.text_id}" 

250 for row in combos.itertuples() 

251 ] 

252 label_to_combo = dict( 

253 zip( 

254 combo_labels, 

255 combos[["participant_id", "trial_id"]].itertuples(index=False, name=None), 

256 ) 

257 ) 

258 return combos, combo_labels, label_to_combo 

259 

260 

261@st.cache_data(show_spinner=False) 

262def _trial_positions(_frame: pd.DataFrame, cache_key) -> dict[tuple[str, str], object]: 

263 """Map ``(participant_id, trial_id)`` → positional row indices. 

264 

265 Built once per frame (cached on its fingerprint) so extracting a single 

266 trial is an O(trial) ``iloc`` rather than an O(corpus) boolean mask on every 

267 rerun — and shared across the tabs, which all slice the same filtered frames. 

268 """ 

269 if _frame is None or _frame.empty: 

270 return {} 

271 grouped = _frame.groupby(["participant_id", "trial_id"], sort=False).indices 

272 # Normalise keys to (str, str) so lookups match the picker's string values. 

273 return {(str(p), str(t)): idx for (p, t), idx in grouped.items()} 

274 

275 

276def extract_trial(frame: pd.DataFrame, participant_id, trial_id) -> pd.DataFrame: 

277 """Rows of one (participant, trial), sliced via the cached position index. 

278 

279 Equivalent to ``frame[(frame.participant_id == p) & (frame.trial_id == t)]`` 

280 but O(trial) instead of O(corpus) once the index is built — the per-rerun win 

281 on large datasets, where every tab extracts the selected trial.""" 

282 if frame is None or getattr(frame, "empty", True): 

283 return frame 

284 positions = _trial_positions(frame, cache_key=frame_fingerprint(frame)) 

285 pos = positions.get((str(participant_id), str(trial_id))) 

286 if pos is None or len(pos) == 0: 

287 return frame.iloc[0:0] 

288 return frame.iloc[pos] 

289 

290 

291# ----------------------------------------------------------------------------- 

292# Trial selection UI 

293# ----------------------------------------------------------------------------- 

294 

295# UX-10 · sorting the trial pool. 

296# 

297# The picker listed trials in data order, so finding "the slowest reader", "the 

298# one with the most fixations" or "the trials this reader got wrong" meant 

299# scrolling the whole list. These build a sort key per trial from three sources: 

300# computed per-trial stats, reader/text properties, and any trial-level column 

301# the dataset carries. Pure and frame-driven, so they're testable without the UI. 

302TRIAL_SORT_DEFAULT = "Trial ID" 

303#: UX-171: the order the trials appear in the data — the picker's default when 

304#: the combos carry it. ``Trial ID`` (sorted by id, so ``1, 10, 100, 2`` for 

305#: numeric ids) stays in the menu as a choice. 

306TRIAL_SORT_DATA_ORDER = "Data order" 

307# Computed stat label → (frame it needs, how to aggregate it per trial). 

308# "fixations" / "words" name which frame the aggregation runs on. 

309# DATA-66: these are the app's, not columns of the dataset, so they say so and 

310# are listed after the dataset's own columns. 

311_TRIAL_SORT_STATS = { 

312 "Fixation count (computed)": ("fixations", "size"), 

313 "Total fixation time, s (computed)": ("fixations", "duration_sum_s"), 

314 "Mean fixation duration, ms (computed)": ("fixations", "duration_mean"), 

315 "Word count (computed)": ("words", "size"), 

316 "First timestamp (computed)": ("fixations", "timestamp_min"), 

317} 

318# Columns worth offering as a sort key when the dataset carries them, in the 

319# order they're shown. Reader properties first, then text, then behaviour. 

320_TRIAL_SORT_PREFERRED_COLS = ( 

321 "participant_id", 

322 "text_id", 

323 "unique_text_id", 

324 "paragraph_id", 

325 "difficulty_level", 

326 "question_preview", 

327 "repeated_reading_trial", 

328 "is_correct", 

329 "genre", 

330 "session", 

331 "pp_age", 

332 "pp_gender", 

333 "TRIAL_INDEX", 

334 "trial_index", 

335) 

336 

337# Event/geometry fields can occasionally be constant by accident (for example a 

338# one-fixation trial), but that does not make them trial metadata. Keep them out 

339# of the generic metadata tail; computed timing/count keys above are the useful 

340# sortable representation of those event columns. 

341_TRIAL_SORT_EXCLUDED_COLS = { 

342 "trial_id", 

343 "unique_trial_id", 

344 "word", 

345 "token", 

346 "text", 

347 "sentence", 

348 "question", 

349 "answer", 

350 "response", 

351 "x", 

352 "y", 

353 "x_start", 

354 "x_end", 

355 "y_start", 

356 "y_end", 

357 "xmin", 

358 "xmax", 

359 "ymin", 

360 "ymax", 

361 "width", 

362 "height", 

363 "timestamp", 

364 "timestamp_ms", 

365 "duration", 

366 "duration_ms", 

367 "order_in_trial", 

368 "fixation_index", 

369 "word_index", 

370 "word_id", 

371 "ia_id", 

372 "char_index", 

373 "line_index", 

374 "source_file", 

375} 

376_TRIAL_SORT_PRIVATE_NAME_PARTS = ( 

377 "path", 

378 "filepath", 

379 "filename", 

380 "directory", 

381 "folder", 

382 "url", 

383 "uri", 

384) 

385_TRIAL_SORT_GEOMETRY_SUFFIXES = ( 

386 "_x", 

387 "_y", 

388 "_xmin", 

389 "_xmax", 

390 "_ymin", 

391 "_ymax", 

392 "_x_start", 

393 "_x_end", 

394 "_y_start", 

395 "_y_end", 

396 "_width", 

397 "_height", 

398) 

399 

400 

401def _trial_sort_column_allowed(column: object) -> bool: 

402 """Whether ``column`` can be discovered as generic trial metadata.""" 

403 name = str(column) 

404 lower = name.lower() 

405 if name.startswith("_") or lower in _TRIAL_SORT_EXCLUDED_COLS: 

406 return False 

407 if lower.endswith(_TRIAL_SORT_GEOMETRY_SUFFIXES): 

408 return False 

409 return not any(part in lower for part in _TRIAL_SORT_PRIVATE_NAME_PARTS) 

410 

411 

412def _effective_trial_field( 

413 frame: pd.DataFrame | None, trial_field: str, picker_ids: set[str] 

414) -> str | None: 

415 """Find the frame column that names the picker's effective trial ids.""" 

416 if frame is None or frame.empty: 

417 return None 

418 for field in (trial_field, "unique_trial_id", "trial_id"): 

419 if field not in frame.columns: 

420 continue 

421 values = set(frame[field].dropna().astype(str).unique()) 

422 if values & picker_ids: 

423 return field 

424 return None 

425 

426 

427def _is_missing_scalar(value) -> bool: 

428 try: 

429 missing = pd.isna(value) 

430 except (TypeError, ValueError): 

431 return False 

432 return bool(missing) if isinstance(missing, (bool, np.bool_)) else False 

433 

434 

435def _same_sort_value(left, right) -> bool: 

436 """Scalar equality for reconciling the words and fixations tables.""" 

437 if _is_missing_scalar(left) or _is_missing_scalar(right): 

438 # Partial missingness is not a conflict; the table carrying a value wins. 

439 return True 

440 try: 

441 equal = left == right 

442 except (TypeError, ValueError): 

443 return False 

444 return bool(equal) if isinstance(equal, (bool, np.bool_)) else False 

445 

446 

447def _looks_like_free_text(series: pd.Series) -> bool: 

448 """Reject prose-like cells while retaining ordinary categorical metadata.""" 

449 values = [str(v).strip() for v in series if not _is_missing_scalar(v)] 

450 return bool(values) and any(len(v) > 200 or len(v.split()) > 24 for v in values) 

451 

452 

453def _trial_level_columns_from_frame( 

454 frame: pd.DataFrame | None, 

455 trial_field: str, 

456 picker_ids: set[str], 

457 participants: set[str], 

458) -> dict[str, pd.Series]: 

459 """Discover one scalar value per active trial directly from one source table. 

460 

461 Grouping includes participant identity, preventing repeated plain trial ids 

462 from being merged before the active picker scope is applied. The returned 

463 Series uses the picker's effective id because that is what 

464 ``sort_trial_options`` consumes. 

465 """ 

466 identity = _effective_trial_field(frame, trial_field, picker_ids) 

467 if frame is None or frame.empty or identity is None: 

468 return {} 

469 

470 scoped = frame 

471 if participants and "participant_id" in scoped.columns: 

472 scoped = scoped[scoped["participant_id"].astype(str).isin(participants)] 

473 scoped = scoped[scoped[identity].astype(str).isin(picker_ids)] 

474 if scoped.empty: 

475 return {} 

476 

477 group_cols = [identity] 

478 if "participant_id" in scoped.columns and identity != "participant_id": 

479 group_cols.insert(0, "participant_id") 

480 grouped = scoped.groupby(group_cols, sort=False, dropna=False) 

481 discovered: dict[str, pd.Series] = {} 

482 for col in scoped.columns: 

483 if col == identity or not _trial_sort_column_allowed(col): 

484 continue 

485 try: 

486 if (grouped[col].nunique(dropna=False) > 1).any(): 

487 continue 

488 if col in group_cols: 

489 values = scoped[group_cols].drop_duplicates().copy() 

490 else: 

491 values = grouped[col].agg(lambda cells: cells.iloc[0]).reset_index() 

492 # If the same effective id survives for multiple participants, it is 

493 # usable only when those rows agree. A participant-narrowed picker 

494 # naturally has one row here; a global ambiguous picker is not 

495 # allowed to choose one participant silently. 

496 by_id = values.groupby(values[identity].astype(str), sort=False)[col] 

497 if (by_id.nunique(dropna=False) > 1).any(): 

498 continue 

499 deduped = values.drop_duplicates(subset=[identity]) 

500 except (TypeError, ValueError): 

501 # Nested/list-like event payloads are not sortable scalar metadata. 

502 continue 

503 series = pd.Series( 

504 deduped[col].to_numpy(), 

505 index=deduped[identity].astype(str).to_numpy(), 

506 ) 

507 if series.dropna().empty or _looks_like_free_text(series): 

508 continue 

509 discovered[str(col)] = series 

510 return discovered 

511 

512 

513def _merge_trial_level_sources( 

514 sources: Iterable[dict[str, pd.Series]], 

515) -> dict[str, pd.Series]: 

516 """Merge compatible metadata sources; omit cross-table disagreements.""" 

517 by_column: dict[str, list[pd.Series]] = {} 

518 for source in sources: 

519 for col, series in source.items(): 

520 by_column.setdefault(col, []).append(series) 

521 

522 merged: dict[str, pd.Series] = {} 

523 for col, series_list in by_column.items(): 

524 combined = pd.Series(dtype=object) 

525 conflict = False 

526 for series in series_list: 

527 current = series.copy() 

528 current.index = current.index.astype(str) 

529 for trial_id in combined.index.intersection(current.index): 

530 if not _same_sort_value(combined[trial_id], current[trial_id]): 

531 conflict = True 

532 break 

533 if conflict: 

534 break 

535 # pandas warns when concatenation/combine_first has to infer a 

536 # dtype from an empty object Series. The first real source needs no 

537 # merge at all; starting from it also preserves its native dtype. 

538 combined = current if combined.empty else combined.combine_first(current) 

539 if not conflict and not combined.empty: 

540 merged[col] = combined 

541 return merged 

542 

543 

544@st.cache_data(show_spinner=False, max_entries=16) 

545def _trial_level_sort_columns_cached( 

546 _combos: pd.DataFrame, 

547 _words: pd.DataFrame | None, 

548 _fixations: pd.DataFrame | None, 

549 trial_field: str, 

550 cache_key, 

551) -> dict[str, pd.Series]: 

552 """Cached metadata discovery over the participant-scoped picker frames.""" 

553 del cache_key # explicit hash input for the underscore-prefixed frames 

554 if _combos is None or _combos.empty or trial_field not in _combos.columns: 

555 return {} 

556 picker_ids = set(_combos[trial_field].dropna().astype(str).unique()) 

557 participants = ( 

558 set(_combos["participant_id"].dropna().astype(str).unique()) 

559 if "participant_id" in _combos.columns 

560 else set() 

561 ) 

562 return _merge_trial_level_sources( 

563 ( 

564 _trial_level_columns_from_frame( 

565 _combos, trial_field, picker_ids, participants 

566 ), 

567 _trial_level_columns_from_frame( 

568 _words, trial_field, picker_ids, participants 

569 ), 

570 _trial_level_columns_from_frame( 

571 _fixations, trial_field, picker_ids, participants 

572 ), 

573 ) 

574 ) 

575 

576 

577def _trial_level_sort_columns( 

578 combos: pd.DataFrame, 

579 trial_field: str, 

580 words: pd.DataFrame | None, 

581 fixations: pd.DataFrame | None, 

582) -> dict[str, pd.Series]: 

583 return _trial_level_sort_columns_cached( 

584 combos, 

585 words, 

586 fixations, 

587 trial_field, 

588 cache_key=( 

589 frame_fingerprint(combos), 

590 frame_fingerprint(words), 

591 frame_fingerprint(fixations), 

592 trial_field, 

593 ), 

594 ) 

595 

596 

597def _per_trial_stat(frame: pd.DataFrame, trial_field: str, how: str) -> pd.Series: 

598 """One computed stat per trial id, as a Series indexed by that id.""" 

599 if frame is None or frame.empty or trial_field not in frame.columns: 

600 return pd.Series(dtype=float) 

601 grouped = frame.groupby(frame[trial_field].astype(str), sort=False) 

602 if how == "size": 

603 return grouped.size().astype(float) 

604 if how == "timestamp_min": 

605 if "timestamp_ms" not in frame.columns: 

606 return pd.Series(dtype=float) 

607 return pd.to_numeric(grouped["timestamp_ms"].min(), errors="coerce").astype( 

608 float 

609 ) 

610 if "duration_ms" not in frame.columns: 

611 return pd.Series(dtype=float) 

612 durations = grouped["duration_ms"].agg("sum" if "sum" in how else "mean") 

613 return (durations / 1000.0) if how.endswith("_s") else durations.astype(float) 

614 

615 

616def trial_sort_keys( 

617 combos: pd.DataFrame, 

618 trial_field: str, 

619 *, 

620 words: pd.DataFrame | None = None, 

621 fixations: pd.DataFrame | None = None, 

622 label_of: Callable[[str], str] = str, 

623) -> dict[str, pd.Series]: 

624 """Available sort keys (UX-10): label → Series indexed by trial id. 

625 

626 The dataset's own trial-level columns come first, each under ``label_of`` 

627 (the dataset's own name, DATA-66 — `ColumnNames.label`), then the statistics 

628 Scanpath Studio computes per trial, marked as computed. 

629 

630 Offers a computed stat only when the frame it needs is present, and a column 

631 only when it is actually trial-level in the active participant-scoped words, 

632 fixations, or combo frame. This deliberately discovers metadata before the 

633 lossy combo projection can discard it. 

634 """ 

635 keys: dict[str, pd.Series] = {} 

636 # This rank was captured before build_combo_options' canonical sort. 

637 if ( 

638 combos is not None 

639 and not combos.empty 

640 and trial_field in combos.columns 

641 and "_data_order" in combos.columns 

642 ): 

643 deduped = combos.drop_duplicates(subset=[trial_field]) 

644 keys[TRIAL_SORT_DATA_ORDER] = pd.Series( 

645 deduped["_data_order"].to_numpy(), 

646 index=deduped[trial_field].astype(str).to_numpy(), 

647 ) 

648 has_combos = ( 

649 combos is not None and not combos.empty and trial_field in combos.columns 

650 ) 

651 if has_combos: 

652 discovered = _trial_level_sort_columns(combos, trial_field, words, fixations) 

653 ordered_cols = [c for c in _TRIAL_SORT_PREFERRED_COLS if c in discovered] 

654 ordered_cols.extend( 

655 sorted(set(discovered) - set(ordered_cols), key=str.casefold) 

656 ) 

657 labelled = [(col, label_of(col)) for col in ordered_cols if col != trial_field] 

658 # The dataset's own columns first, then the ones the app made. 

659 labelled.sort(key=lambda pair: pair[1].endswith(COMPUTED_SUFFIX)) 

660 for col, label in labelled: 

661 series = discovered[col] 

662 if label in keys: 

663 # An alias read from the same column sorts the same way: once. 

664 if keys[label].equals(series): 

665 continue 

666 label = f"{label} ({col})" 

667 elif label in (TRIAL_SORT_DEFAULT, TRIAL_SORT_DATA_ORDER): 

668 # A column the dataset itself calls "Trial ID" is not the menu's. 

669 label = f"{label} ({col})" 

670 keys[label] = series 

671 keys.update( 

672 _trial_sort_stats_cached( 

673 combos if has_combos else None, 

674 words, 

675 fixations, 

676 trial_field, 

677 cache_key=( 

678 frame_fingerprint(combos) if has_combos else None, 

679 frame_fingerprint(words), 

680 frame_fingerprint(fixations), 

681 trial_field, 

682 ), 

683 ) 

684 ) 

685 return keys 

686 

687 

688@st.cache_data(show_spinner=False, max_entries=16) 

689def _trial_sort_stats_cached( 

690 _combos: pd.DataFrame | None, 

691 _words: pd.DataFrame | None, 

692 _fixations: pd.DataFrame | None, 

693 trial_field: str, 

694 cache_key, 

695) -> dict[str, pd.Series]: 

696 """The computed sort keys (fixation count, reading time …), label → Series. 

697 

698 Each is a group-by over the whole fixation or word table, ~0.25 s a rerun 

699 at OneStop scale for numbers that change only with the trial pool.""" 

700 del cache_key # explicit hash input for the underscore-prefixed frames 

701 picker_ids = ( 

702 set(_combos[trial_field].dropna().astype(str).unique()) 

703 if _combos is not None 

704 else set() 

705 ) 

706 stats: dict[str, pd.Series] = {} 

707 for label, (which, how) in _TRIAL_SORT_STATS.items(): 

708 frame = _fixations if which == "fixations" else _words 

709 field = _effective_trial_field(frame, trial_field, picker_ids) 

710 series = _per_trial_stat(frame, field or trial_field, how) 

711 if not series.empty: 

712 stats[label] = series 

713 return stats 

714 

715 

716def sort_trial_options( 

717 options: list[str], 

718 key_series: pd.Series | None, 

719 *, 

720 descending: bool = False, 

721) -> list[str]: 

722 """Order ``options`` (trial ids) by ``key_series``, ties broken by id. 

723 

724 Trials the key doesn't cover sort last regardless of direction — an unranked 

725 trial is missing information, not an extreme value, so it shouldn't lead. 

726 """ 

727 if key_series is None or key_series.empty: 

728 return sorted(options) 

729 lookup = key_series.to_dict() 

730 ranked = [o for o in options if o in lookup and pd.notna(lookup[o])] 

731 ranked_set = set(ranked) 

732 unranked = sorted(o for o in options if o not in ranked_set) 

733 ranked.sort(key=lambda o: (_sort_scalar(lookup[o]), o), reverse=descending) 

734 return ranked + unranked 

735 

736 

737def _sort_scalar(value): 

738 """A comparable key for a cell that may be numeric, boolean or text.""" 

739 if isinstance(value, bool): 

740 return (0, float(value)) 

741 try: 

742 return (0, float(value)) 

743 except (TypeError, ValueError): 

744 return (1, str(value)) 

745 

746 

747def format_sort_value(value) -> str: 

748 """A sort key's value, short enough to ride along in a picker option. 

749 

750 Sorting the pool is only useful if you can *see* what you sorted by — an 

751 ordering with the ordering key hidden just looks shuffled. Integers keep a 

752 thousands separator, floats get one decimal, booleans read Yes/No. 

753 """ 

754 if value is None or (isinstance(value, float) and pd.isna(value)): 

755 return "—" 

756 if isinstance(value, (bool, np.bool_)): 

757 return "Yes" if value else "No" 

758 if isinstance(value, (int, float, np.integer, np.floating)): 

759 number = float(value) 

760 return f"{number:,.0f}" if number == int(number) else f"{number:,.1f}" 

761 return str(value) 

762 

763 

764def _trial_display_label(trial_id) -> str: 

765 """Human-readable label for a trial id in the pickers. 

766 

767 A per-page trial id reads cleanly — ``Lit_Alchemist_4__page_07`` → 

768 ``Lit_Alchemist_4 · page 7`` (the id stays zero-padded so it sorts 

769 numerically; only the display drops the padding). Any other id passes 

770 through unchanged, so this is a no-op for every other corpus. MultiplEYE 

771 used to be the one producing those ids; since DATA-24 its pages are screens 

772 inside one trial, so this is now generic dressing for whatever ships them.""" 

773 text = str(trial_id) 

774 stim, sep, page = text.rpartition("__page_") 

775 if sep and page.isdigit(): 

776 return f"{stim} · page {int(page)}" 

777 return text 

778 

779 

780# --- UX-187: a trial id spelled out part by part ------------------------------ 

781# A trial id is usually several ids joined with "_" — OneStop's 

782# `l37_1129_2_2_1_Adv_r0` is reader `l37_1129`, text `2_2_1_Adv`, first reading. 

783# The pickers show it with " · " between the parts so you can tell where one 

784# ends, which a plain split on "_" cannot do: the reader id has an underscore of 

785# its own. So the parts are found from the ids the trial is known to be made of — 

786# its participant, its text, a composite mapping's columns — and an id that 

787# matches none of them is shown exactly as it is. Only the display changes: the 

788# selection, the deep link and every export keep the id itself. 

789# 

790# UX-202: the text id is one part, as it is everywhere else in the app (it was 

791# split into OneStop's batch · article · paragraph · level), and a first reading 

792# (`r0`) is not shown — only a repeated one says which reading it is. 

793 

794#: What the pickers put between the parts of a trial id. 

795TRIAL_ID_PART_SEPARATOR = " · " 

796 

797#: UX-202 — the reading number a first reading carries, which the display leaves 

798#: out. OneStop composes `_r0` / `_r1`; `data._disambiguate_repeated_readings` 

799#: leaves a first reading unsuffixed and numbers the next `_r2`. 

800FIRST_READING = "r0" 

801 

802_READING_SUFFIX = re.compile(r"_(r\d+)$") 

803 

804 

805def trial_id_parts( 

806 trial_id, 

807 *, 

808 participant_id=None, 

809 text_id=None, 

810 components: list[tuple[str, str]] | None = None, 

811) -> list[tuple[str, str]]: 

812 """``(name, value)`` for each part of ``trial_id``, in the order it is written. 

813 

814 ``components`` (a composite trial mapping's ``(column, value)`` pairs) are the 

815 answer outright. Otherwise the id is read as ``[<participant>_]<text>[_rN]``. 

816 An id that shape does not account for entirely is one part, 

817 ``("trial id", trial_id)``. 

818 """ 

819 if components: 

820 return [(str(name), str(value)) for name, value in components] 

821 tid = str(trial_id) 

822 whole = [("trial id", tid)] 

823 pid = "" if participant_id is None else str(participant_id) 

824 text = "" if text_id is None else str(text_id) 

825 parts: list[tuple[str, str]] = [] 

826 rest = tid 

827 if pid and rest.startswith(f"{pid}_"): 

828 parts.append(("participant", pid)) 

829 rest = rest[len(pid) + 1 :] 

830 if text and text != tid and (rest == text or rest.startswith(f"{text}_")): 

831 parts.append(("text", text)) 

832 rest = rest[len(text) + 1 :] 

833 # Split only an id its parts account for entirely: the text must be in it, 

834 # and anything after the text a reading number. `synthetic_2line_demo` 

835 # starts with its reader's id, but is not made of it. 

836 if not parts or parts[-1][0] == "participant": 

837 return whole 

838 if rest: 

839 if not (rest[:1] == "r" and rest[1:].isdigit()): 

840 return whole 

841 parts.append(("reading", rest)) 

842 return parts 

843 

844 

845def shown_parts(parts: list[tuple[str, str]]) -> list[tuple[str, str]]: 

846 """The parts the display writes: all but a first reading (UX-202).""" 

847 return [ 

848 (name, value) 

849 for name, value in parts 

850 if not (name == "reading" and value == FIRST_READING) 

851 ] 

852 

853 

854def trial_id_display(parts: list[tuple[str, str]]) -> str: 

855 """A trial id as the pickers show it: its parts, `TRIAL_ID_PART_SEPARATOR`-joined. 

856 

857 A first reading is left out (UX-202), and an id that does not split still 

858 has a trailing reading number set off by the separator, not an underscore. 

859 """ 

860 parts = shown_parts(parts) 

861 if len(parts) == 1: 

862 label = _trial_display_label(parts[0][1]) 

863 return _READING_SUFFIX.sub(rf"{TRIAL_ID_PART_SEPARATOR}\1", label) 

864 return TRIAL_ID_PART_SEPARATOR.join(value for _, value in parts) 

865 

866 

867def trial_id_layout( 

868 combos: pd.DataFrame, 

869 trial_field: str = "trial_id", 

870 *, 

871 composite_cols: Iterable[str] = (), 

872) -> tuple[dict[str, str], tuple[str, ...]]: 

873 """Display string per trial id in ``combos``, plus the part names they share. 

874 

875 ``composite_cols`` are the composite trial mapping's columns (carried on 

876 ``combos``). The part names are the most common shown layout's — what the 

877 picker's help names — with ``"reading"`` added when any trial shows one, and 

878 empty when no id splits. 

879 """ 

880 if combos.empty or trial_field not in combos.columns: 

881 return {}, () 

882 composite = [c for c in composite_cols if c in combos.columns] 

883 text_field = next( 

884 (c for c in ("unique_text_id", "text_id") if c in combos.columns), None 

885 ) 

886 rows = combos.drop_duplicates(subset=[trial_field]) 

887 trial_ids = rows[trial_field].astype(str).to_numpy() 

888 pids = rows["participant_id"].to_numpy() if "participant_id" in rows else None 

889 texts = rows[text_field].to_numpy() if text_field else None 

890 comp_values = ( 

891 rows[composite].apply(stable_id).to_numpy() if len(composite) > 1 else None 

892 ) 

893 display: dict[str, str] = {} 

894 layouts: dict[tuple[str, ...], int] = {} 

895 any_reading = False 

896 for i, tid in enumerate(trial_ids): 

897 parts = trial_id_parts( 

898 tid, 

899 participant_id=None if pids is None else pids[i], 

900 text_id=None if texts is None else texts[i], 

901 components=( 

902 None 

903 if comp_values is None 

904 else list(zip(composite, comp_values[i], strict=True)) 

905 ), 

906 ) 

907 display[tid] = trial_id_display(parts) 

908 shown = shown_parts(parts) 

909 if len(shown) > 1: 

910 names = tuple(name for name, _ in shown if name != "reading") 

911 layouts[names] = layouts.get(names, 0) + 1 

912 any_reading = any_reading or any(name == "reading" for name, _ in shown) 

913 names = max(layouts, key=layouts.__getitem__) if layouts else () 

914 if names and any_reading: 

915 names = (*names, "reading") 

916 return display, names 

917 

918 

919def trial_id_shown( 

920 trial_id, 

921 *frames: pd.DataFrame | None, 

922 participant_id=None, 

923 composite_cols: Iterable[str] = (), 

924) -> str: 

925 """One trial's id as the pickers show it, read off the first of ``frames`` 

926 (that trial's own rows) that has any. ``participant_id`` overrides the 

927 frame's — a cross-dataset B carries a namespaced one (`qualify_for_compare`).""" 

928 frame = next((f for f in frames if f is not None and not f.empty), None) 

929 if frame is None: 

930 return trial_id_display(trial_id_parts(trial_id)) 

931 row = frame.iloc[:1].copy() 

932 row["trial_id"] = str(trial_id) 

933 if participant_id is not None: 

934 row["participant_id"] = participant_id 

935 display, _ = trial_id_layout(row, composite_cols=composite_cols) 

936 return display.get(str(trial_id), str(trial_id)) 

937 

938 

939def trial_id_help(part_names: tuple[str, ...]) -> str: 

940 """The sentence the pickers' help gives about what a trial id is made of.""" 

941 if not part_names: 

942 return "" 

943 sentence = ( 

944 "A trial id is written as its parts, " 

945 f"**{TRIAL_ID_PART_SEPARATOR.join(part_names)}**." 

946 ) 

947 if "reading" in part_names: 

948 sentence += ( 

949 " A repeated reading ends in its number (`r1`, `r2` …); a first" 

950 " reading has none." 

951 ) 

952 return sentence 

953 

954 

955def _render_trial_sort_popover( 

956 host, 

957 combos: pd.DataFrame, 

958 trial_field: str, 

959 key_prefix: str, 

960 *, 

961 words: pd.DataFrame | None, 

962 fixations: pd.DataFrame | None, 

963) -> tuple[pd.Series | None, bool, str]: 

964 """The ⇅ sort control beside the trial picker (UX-10). 

965 

966 Lives in a popover rather than inline: the picker row is already a selectbox, 

967 a slider and two step buttons wide, and sorting is a "set it once" choice, not 

968 a per-trial one. Returns ``(key_series, descending, choice)`` for 

969 :func:`sort_trial_options` — ``(None, False, TRIAL_SORT_DEFAULT)`` for the 

970 default id order. The chosen key's *name* comes back too, because the picker 

971 labels the ordering it's showing. 

972 """ 

973 keys = trial_sort_keys( 

974 combos, 

975 trial_field, 

976 words=words, 

977 fixations=fixations, 

978 label_of=active_all(st.session_state).label, 

979 ) 

980 if not keys: 

981 return None, False, TRIAL_SORT_DEFAULT 

982 # UX-171: data order leads and is the default; Trial ID follows it. 

983 default = TRIAL_SORT_DATA_ORDER if TRIAL_SORT_DATA_ORDER in keys else None 

984 options = [ 

985 *([default] if default else []), 

986 TRIAL_SORT_DEFAULT, 

987 *(k for k in keys if k != default), 

988 ] 

989 state_key = f"{key_prefix}_trial_sort" 

990 if st.session_state.get(state_key) not in options: 

991 st.session_state[state_key] = options[0] 

992 # UX-200: named for screen readers; `styles.py` draws ⇅ alone. 

993 with host.popover( 

994 "Sort the trial list", 

995 width="content", 

996 wrap=True, 

997 help="Sort the trial list", 

998 key=f"iconpop_sort_trial_{key_prefix}", 

999 ): 

1000 choice = labeled( 

1001 st, 

1002 "selectbox", 

1003 "Sort trials by", 

1004 options=options, 

1005 key=state_key, 

1006 help="Reorder the trial list by a computed statistic or by a participant, " 

1007 "text or condition property.", 

1008 ) 

1009 descending = labeled( 

1010 st, 

1011 "checkbox", 

1012 "Descending", 

1013 key=f"{key_prefix}_trial_sort_desc", 

1014 help="Reverse the order.", 

1015 ) 

1016 if choice == TRIAL_SORT_DEFAULT: 

1017 return None, False, TRIAL_SORT_DEFAULT 

1018 return keys[choice], bool(descending), choice 

1019 

1020 

1021# --- CMP-13: one ◀ ▶ that advances both compared trials ---------------------- 

1022# The two pickers are built in different modules (A here, B in `tabs.py`), so the 

1023# linked step is a callback on one side writing the *other* side's selection. It 

1024# needs the other list, which is why each picker publishes what it just rendered. 

1025# Both directions clamp independently: per the settled call, a side that has run 

1026# out simply stays put while the other keeps stepping. 

1027 

1028#: The ⚙️ Compare options checkbox that arms the link. UI-only — a navigation 

1029#: control, not a render setting, so it is deliberately not on the share link or 

1030#: in a saved config (same call as ``share_identity_mode``). 

1031COMPARE_STEP_LINK_KEY = "single_compare_step_linked" 

1032 

1033#: Scanpath B's canonical selection (a *label*; see the snapshot note below). 

1034COMPARE_TRIAL_KEY = "single_compare_trial" 

1035 

1036#: What the *Compare To* picker last rendered: ``(label, participant, trial)`` 

1037#: per candidate, in display order. 

1038COMPARE_OPTIONS_SNAPSHOT_KEY = "_compare_options_snapshot" 

1039 

1040 

1041def trial_options_snapshot_key(key_prefix: str) -> str: 

1042 """Session key holding the trial picker's options as it last rendered them.""" 

1043 return f"_{key_prefix}_trial_options" if key_prefix else "_trial_options" 

1044 

1045 

1046def compare_step_linked() -> bool: 

1047 """True when ◀ ▶ should advance scanpath A **and** B (CMP-13). 

1048 

1049 Both halves matter: the checkbox only exists while compare mode is on, and 

1050 Streamlit drops an unrendered widget's key, so a stale ``True`` must not 

1051 quietly steer the main picker once the user has left compare mode. 

1052 """ 

1053 return bool( 

1054 st.session_state.get("single_compare_toggle") 

1055 and st.session_state.get(COMPARE_STEP_LINK_KEY) 

1056 ) 

1057 

1058 

1059def step_within(options: list[str], state_key: str, delta: int) -> int | None: 

1060 """Move ``state_key``'s selection ``delta`` places within ``options``. 

1061 

1062 Clamped to the ends, and clamped *independently* of any other picker — the 

1063 linked step is "advance both", not "keep them aligned": the two pools have 

1064 different sizes (B has its own filters, and a cross-dataset B is another 

1065 corpus entirely), so their indices carry no shared meaning. 

1066 

1067 Returns the new index, or ``None`` when there was nothing to step. 

1068 """ 

1069 opts = list(options or []) 

1070 if not opts: 

1071 return None 

1072 try: 

1073 pos = opts.index(st.session_state.get(state_key)) 

1074 except ValueError: 

1075 pos = 0 

1076 new_pos = max(0, min(pos + delta, len(opts) - 1)) 

1077 st.session_state[state_key] = opts[new_pos] 

1078 return new_pos 

1079 

1080 

1081def at_list_end(options: list[str], state_key: str, delta: int) -> bool: 

1082 """True when ``state_key``'s selection cannot move ``delta`` within ``options``. 

1083 

1084 Used to decide whether a step button is dead. An unknown list answers 

1085 **False** so the button stays live: a click that turns out to be a no-op is a 

1086 better failure than a button greyed out while the other side could still move. 

1087 """ 

1088 opts = list(options or []) 

1089 if not opts: 

1090 return False 

1091 try: 

1092 pos = opts.index(st.session_state.get(state_key)) 

1093 except ValueError: 

1094 return False 

1095 return not (0 <= pos + delta < len(opts)) 

1096 

1097 

1098def step_linked_compare(delta: int) -> None: 

1099 """Advance scanpath **B** by ``delta``, resolved against the list A last saw. 

1100 

1101 Written as an *identity* rather than an index or a label, because both are 

1102 unstable across this step: ``build_comparison_options`` builds B's pool 

1103 relative to A (📄 same-text first, then 👤 same-participant), so once A 

1104 moves, B's list is re-ordered *and* re-labelled — the same trial can gain or 

1105 lose its 📄 marker. Parking the identity in the same 

1106 pending slot the ``?compare=`` deep link uses lets the rebuilt picker re-find 

1107 the trial the user was actually looking at. 

1108 """ 

1109 from .session_keys import PENDING_COMPARE_STATE_KEY 

1110 

1111 snapshot = list(st.session_state.get(COMPARE_OPTIONS_SNAPSHOT_KEY) or []) 

1112 if not snapshot: 

1113 return 

1114 labels = [row[0] for row in snapshot] 

1115 try: 

1116 pos = labels.index(st.session_state.get(COMPARE_TRIAL_KEY)) 

1117 except ValueError: 

1118 pos = 0 

1119 _, participant, trial = snapshot[max(0, min(pos + delta, len(snapshot) - 1))] 

1120 st.session_state[PENDING_COMPARE_STATE_KEY] = { 

1121 "participant_id": participant, 

1122 "trial_id": trial, 

1123 } 

1124 

1125 

1126#: #374 F34 — ``{"<key prefix>|<dataset>": trial id}``: the trial each 

1127#: dataset's picker was last on. 

1128_TRIAL_BY_DATASET_KEY = "_trial_by_dataset" 

1129 

1130 

1131def row_tail(column, key_prefix: str, reserve_screen_cell): 

1132 """Lay out a selector row's last cell: the screen navigator, then the menus. 

1133 

1134 ``reserve_screen_cell`` is handed the cell's container to keep a slot for 

1135 the screen navigator (filled once the trial is resolved). Returns the 

1136 ``railbtn_*`` cluster after it, where the row's ⇅ 🔎 ✏️ go — at the row's 

1137 right end, while ◀ ▶ stay beside the slider they step. 

1138 """ 

1139 tail = column.container( 

1140 key=f"{key_prefix}_row_tail", 

1141 horizontal=True, 

1142 vertical_alignment="bottom", 

1143 gap="small", 

1144 ) 

1145 reserve_screen_cell(tail) 

1146 return tail.container(key=f"railbtn_{key_prefix}_menus", width="content") 

1147 

1148 

1149def _select_trial_none_mode( 

1150 combos: pd.DataFrame, 

1151 trial_field: str, 

1152 text_field: str, 

1153 key_prefix: str, 

1154 picker_host=None, 

1155 *, 

1156 words: pd.DataFrame | None = None, 

1157 fixations: pd.DataFrame | None = None, 

1158 leading_renderer=None, 

1159 filter_renderer=None, 

1160 trailing_renderer=None, 

1161) -> tuple[str | None, str | None, str | None]: 

1162 """The trial picker: **dataset + selectbox + scrubbing slider + ◀ ▶ steps + ⇅ 

1163 sort + 🔎 filters**, all on one row (UX-64). The slider thumb shows ``index/TOTAL · id`` 

1164 (index first). The pool is narrowed upstream (the "Narrow by" multiselects + 

1165 the "More" filters), so this just picks one trial from it and orders it. 

1166 

1167 Creates its own row of columns, so call it where columns are allowed (the 

1168 Scanpath/Corpus body), not nested inside another column. ``picker_host`` 

1169 (when given) is the container to render into; defaults to the current one. 

1170 ``words`` / ``fixations`` (optional) unlock the computed sort keys (UX-10); 

1171 without them only column-based orderings are offered. 

1172 

1173 ``trailing_renderer`` (optional) is handed a slot in one more column at 

1174 the row's right end — the multipart screen navigator, which the caller can 

1175 only draw once the trial is resolved, so it keeps the slot and fills it 

1176 later. The row's ⇅ 🔎 ✏️ then move after it (``row_tail``).""" 

1177 host = picker_host if picker_host is not None else st 

1178 available_trials = combos.drop_duplicates(subset=[trial_field]) 

1179 trial_options = sorted(available_trials[trial_field].dropna().astype(str).unique()) 

1180 if not trial_options: 

1181 st.warning( 

1182 "No trials match the filters. Clear one, or use ✕ Clear all filters." 

1183 ) 

1184 st.stop() 

1185 

1186 # Trial id → participant, so the annotation markers (UX-6) can be looked up per 

1187 # option (annotations are keyed by (participant, trial)). Mirrors the selection 

1188 # below, which resolves the participant the same way (first matching row). 

1189 trial_to_pid = dict( 

1190 zip( 

1191 available_trials[trial_field].astype(str), 

1192 available_trials["participant_id"], 

1193 strict=True, 

1194 ) 

1195 ) 

1196 

1197 # Populated once the ⇅ popover has rendered (below), and read by the option 

1198 # labels — so an active ordering is *visible* in the picker itself rather than 

1199 # only inside the popover that set it. 

1200 sort_values: dict[str, str] = {} 

1201 

1202 # UX-187: each id shown part by part, and the help says what the parts are. 

1203 id_display, id_part_names = trial_id_layout( 

1204 available_trials, 

1205 trial_field, 

1206 composite_cols=st.session_state.get("_composite_trial_columns") or (), 

1207 ) 

1208 

1209 # Read once: a picker lists every trial in the pool, and going through the 

1210 # session for each one cost ~0.3 s a rerun at OneStop scale. 

1211 store = store_for_prefix() 

1212 

1213 def _option_label(value: str) -> str: 

1214 marks = annotation_markers(trial_to_pid.get(value), value, store=store) 

1215 base = id_display.get(value) or _trial_display_label(value) 

1216 # #374 F27: the badges follow the trial, so a narrow picker cuts the 

1217 # badges rather than the trial. 

1218 label = f"{base} {marks}" if marks else base 

1219 shown = sort_values.get(value) 

1220 return f"{label} · {shown}" if shown else label 

1221 

1222 # Every label made once, when the order is final (filled below): Streamlit 

1223 # formats each option more than once a run, and the slider's twice over. 

1224 option_labels: dict[str, str] = {} 

1225 

1226 def _format_option(value: str) -> str: 

1227 label = option_labels.get(value) 

1228 return _option_label(value) if label is None else label 

1229 

1230 n_trials = len(trial_options) 

1231 picker_label = "Select trial" 

1232 trial_id_key = f"{key_prefix}_trial_id" if key_prefix else None 

1233 slider_key = f"{key_prefix}_trial_pos" if key_prefix else "trial_pos" 

1234 

1235 # The selectbox (`*_trial_id`) is the canonical selection — the deep-link / 

1236 # Save-&-restore code seeds it (`_restore_selection`). The slider mirrors it 

1237 # and ◀ ▶ step it; all stay in sync via the trial id. 

1238 current_label = st.session_state.get(trial_id_key) if trial_id_key else None 

1239 # #374 F34: the trial each dataset was last on, so switching away and back 

1240 # returns to it rather than to the first trial. 

1241 dataset = annotations_dataset(st.session_state) 

1242 remembered = st.session_state.setdefault(_TRIAL_BY_DATASET_KEY, {}) 

1243 last_dataset_key = f"{_TRIAL_BY_DATASET_KEY}_{key_prefix}" 

1244 previous = st.session_state.get(last_dataset_key, dataset) 

1245 st.session_state[last_dataset_key] = dataset 

1246 # A trial carried over from the dataset left behind is not a choice; one a 

1247 # link or a restored settings file put there with the switch is. 

1248 chosen = st.session_state.pop(f"_{key_prefix}_trial_chosen", None) 

1249 carried = ( 

1250 previous != dataset 

1251 and current_label != chosen 

1252 and current_label == remembered.get(f"{key_prefix}|{previous}") 

1253 ) 

1254 back_to = remembered.get(f"{key_prefix}|{dataset}") 

1255 if ( 

1256 trial_id_key 

1257 and back_to in trial_options 

1258 and (carried or current_label not in trial_options) 

1259 ): 

1260 current_label = back_to 

1261 st.session_state[trial_id_key] = current_label 

1262 # Seeded rather than chosen: re-seeded to the *sorted* list's first trial 

1263 # once the ⇅ order is known (UX-171 — data order's first, not the id's). 

1264 seeded = current_label not in trial_options 

1265 if seeded: 

1266 current_label = trial_options[0] 

1267 if trial_id_key: 

1268 st.session_state[trial_id_key] = current_label 

1269 

1270 idx_of = {opt: i for i, opt in enumerate(trial_options)} 

1271 

1272 if n_trials > 1: 

1273 # Mirror the slider to the current selection BEFORE it renders, so picking 

1274 # a trial in the dropdown moves the slider too; the drag callback writes 

1275 # the selectbox key, reconciling next run. 

1276 st.session_state[slider_key] = current_label 

1277 

1278 def _on_trial_slider() -> None: 

1279 if trial_id_key: 

1280 st.session_state[trial_id_key] = st.session_state[slider_key] 

1281 

1282 def _step_trial(delta: int) -> None: 

1283 # ◀ / ▶ : move the canonical selection one trial earlier/later. 

1284 if not trial_id_key: 

1285 return 

1286 step_within(trial_options, trial_id_key, delta) 

1287 # CMP-13: while the link is armed, the same ±1 also moves scanpath B. 

1288 if compare_step_linked(): 

1289 step_linked_compare(delta) 

1290 

1291 def _slider_label(value: str) -> str: 

1292 # index/TOTAL first, then the id — the slider doubles as the counter, 

1293 # so a separate "Trial X / N" caption is redundant. 

1294 return f"{idx_of.get(value, 0) + 1}/{n_trials} · {_format_option(value)}" 

1295 

1296 # UX-64 — ONE row for everything: [dataset] [trial] [slider] [◀ ▶ ⇅ 🔎]. 

1297 # The Narrow-by row above it is gone; its filters live in the 🔎 popover 

1298 # at the end of this row. The dataset picker keeps its width on purpose 

1299 # (you may be comparing two datasets, and the label is what tells them 

1300 # apart) — the slider gives up the room instead, and the filter is an 

1301 # icon, which is what makes six controls fit. 

1302 # 

1303 # The three triggers share ONE trailing column as a `railbtn_*` cluster 

1304 # (UX-27), which styles.py packs right at a uniform 3px spacing. A column 

1305 # each put a full gutter between them, so a prev/next *pair* didn't read 

1306 # as a pair. 

1307 lead_col, sel_col, slider_col, trail_col, *extra = host.columns( 

1308 SELECTOR_ROW_GRID + ([SELECTOR_SCREEN_TRACK] if trailing_renderer else []), 

1309 vertical_alignment="bottom", 

1310 ) 

1311 if leading_renderer is not None: 

1312 leading_renderer(lead_col) 

1313 menus = ( 

1314 row_tail(extra[0], key_prefix, trailing_renderer) 

1315 if trailing_renderer is not None 

1316 else None 

1317 ) 

1318 trail = trail_col.container( 

1319 key=f"railbtn_{key_prefix}_trail{'_steps' if menus else ''}" 

1320 ) 

1321 # Created in display order (◀ ▶ then ⇅) but filled out of order: the sort 

1322 # popover has to render first, because the order it returns is what the 

1323 # selectbox, the slider and the ◀ ▶ steps all walk. With a screen cell, 

1324 # ⇅ 🔎 ✏️ close the row after it, and ◀ ▶ stay by the slider. 

1325 step_col = trail.container(key=f"railbtn_{key_prefix}_step") 

1326 cluster = trail if menus is None else menus 

1327 sort_col = cluster.container(key=f"railbtn_{key_prefix}_sort") 

1328 filter_col = cluster.container(key=f"railbtn_{key_prefix}_filter") 

1329 # Filled by the caller, which owns the filter widgets — but created here, 

1330 # in display order, so 🔎 lands after ⇅ in the cluster (UX-64). 

1331 if filter_renderer is not None: 

1332 filter_renderer(filter_col) 

1333 # UX-10: order the pool *before* the widgets read `trial_options`, so the 

1334 # selectbox, the slider and the ◀ ▶ steps all walk the same order. The 

1335 # canonical selection is a trial *id*, so re-sorting never changes which 

1336 # trial is selected — only where it sits in the list. 

1337 # UX-27: each of the three step/sort triggers goes in a `railbtn_*` 

1338 # container so styles.py can give the whole cluster above the plot one 

1339 # button shape — these were square (each `width="stretch"` inside its own 

1340 # narrow column) beside pill-shaped labelled buttons on the rows above 

1341 # and below. 

1342 sort_key, sort_desc, sort_choice = _render_trial_sort_popover( 

1343 sort_col, 

1344 combos, 

1345 trial_field, 

1346 key_prefix, 

1347 words=words, 

1348 fixations=fixations, 

1349 ) 

1350 if sort_key is not None: 

1351 trial_options = sort_trial_options( 

1352 trial_options, sort_key, descending=sort_desc 

1353 ) 

1354 idx_of = {opt: i for i, opt in enumerate(trial_options)} 

1355 # UX-171: data order is the default and its values are bare ranks, 

1356 # so it carries no per-option value and names itself only reversed. 

1357 if sort_choice != TRIAL_SORT_DATA_ORDER: 

1358 lookup = sort_key.to_dict() 

1359 sort_values.update( 

1360 {opt: format_sort_value(lookup.get(opt)) for opt in trial_options} 

1361 ) 

1362 if sort_choice != TRIAL_SORT_DATA_ORDER or sort_desc: 

1363 picker_label = ( 

1364 f"Select trial · by {sort_choice} {'↓' if sort_desc else '↑'}" 

1365 ) 

1366 if seeded: 

1367 current_label = trial_options[0] 

1368 if trial_id_key: 

1369 st.session_state[trial_id_key] = current_label 

1370 st.session_state[slider_key] = current_label 

1371 current_idx = trial_options.index(current_label) 

1372 else: 

1373 # A one-trial pool has no slider (`st.select_slider` throws on a single 

1374 # option — BUG-23) and nothing to step through, but it still needs the 

1375 # dataset picker and the filters: a pool of one is *usually the result of 

1376 # a filter*, so this is exactly when the user reaches for them. UX-64. 

1377 lead_col, sel_col, trail_col, *extra = host.columns( 

1378 SELECTOR_ROW_TRIO + ([SELECTOR_SCREEN_TRACK] if trailing_renderer else []), 

1379 vertical_alignment="bottom", 

1380 ) 

1381 if leading_renderer is not None: 

1382 leading_renderer(lead_col) 

1383 menus = ( 

1384 row_tail(extra[0], key_prefix, trailing_renderer) 

1385 if trailing_renderer is not None 

1386 else None 

1387 ) 

1388 if filter_renderer is not None: 

1389 filter_renderer( 

1390 (trail_col if menus is None else menus).container( 

1391 key=f"railbtn_{key_prefix}_filter_solo" 

1392 ) 

1393 ) 

1394 

1395 # CMP-13: publish the list as rendered (post-sort), so the *Compare To* 

1396 # picker's linked ◀ ▶ can step this picker without rebuilding its ordering. 

1397 if trial_id_key: 

1398 st.session_state[trial_options_snapshot_key(key_prefix)] = list(trial_options) 

1399 # #374 F10: the browser identifies the picked option by its *label*, and 

1400 # a label carries the trial's ★ 🏷️ 📝 marks. Written only when it moved, 

1401 # the browser kept the label from that run; a tag added later renamed the 

1402 # option, the old label matched nothing, and the view fell back to trial 

1403 # 1. Written every run (as the slider is), the browser always holds the 

1404 # label it was last shown, which the next run can still read back. 

1405 st.session_state[trial_id_key] = current_label 

1406 remembered[f"{key_prefix}|{dataset}"] = current_label 

1407 

1408 option_labels.update({opt: _option_label(opt) for opt in trial_options}) 

1409 

1410 # The label is shown so its help "?" icon (the type-to-search hint) is visible 

1411 # — a collapsed label hides it. 

1412 selected_trial_label = sel_col.selectbox( 

1413 picker_label, 

1414 options=trial_options, 

1415 key=trial_id_key, 

1416 # BUG-80: the picker renders only on the Scanpath view, and Streamlit 

1417 # drops an unrendered widget's key at the end of the run — so a trip to 

1418 # Corpus Analysis or 🗂️ Data came back on trial 1 (and, in Compare, 

1419 # left A and B on different texts). A value the pool no longer holds is 

1420 # still reset above, before this renders. 

1421 persist_state="session", 

1422 format_func=_format_option, 

1423 help=" ".join( 

1424 filter( 

1425 None, 

1426 ( 

1427 trial_id_help(id_part_names), 

1428 "Click this dropdown, then type to narrow the list. " 

1429 "★ favorite · 🏷️ tagged · 📝 has notes. When a sort key is " 

1430 "active, each option ends with that trial's value for it.", 

1431 ), 

1432 ) 

1433 ), 

1434 ) 

1435 if trial_id_key: 

1436 # The menu opens as wide as its longest trial id (+ marks / sort value). 

1437 widen_menu(trial_id_key, option_labels.values()) 

1438 

1439 if n_trials > 1: 

1440 slider_labels = {opt: _slider_label(opt) for opt in trial_options} 

1441 with slider_col: 

1442 st.select_slider( 

1443 "Trial", 

1444 options=trial_options, 

1445 key=slider_key, 

1446 persist_state="session", 

1447 on_change=_on_trial_slider, 

1448 help=f"Scrub through the {n_trials} trials (index/total · id, " 

1449 "plus the sort value when one is active); the dropdown jumps to " 

1450 "a specific id.", 

1451 label_visibility="collapsed", 

1452 format_func=lambda value: ( 

1453 slider_labels.get(value) or _slider_label(value) 

1454 ), 

1455 ) 

1456 # Both step buttons in the keyed container reserved above, which styles.py 

1457 # lays out as a flex ROW (a Streamlit vertical block stacks its children 

1458 # by default). 

1459 steps = step_col 

1460 # CMP-13: linked, a button stays live until BOTH sides have run out — a 

1461 # side at the end of its own list just stays put while the other keeps 

1462 # stepping. `at_list_end` answers False for a list it can't see, so the 

1463 # worst case is a click that moves only one scanpath. 

1464 linked = compare_step_linked() 

1465 compare_snapshot = ( 

1466 [ 

1467 row[0] 

1468 for row in (st.session_state.get(COMPARE_OPTIONS_SNAPSHOT_KEY) or []) 

1469 ] 

1470 if linked 

1471 else [] 

1472 ) 

1473 step_help = " Linked: also steps the compared trial." if linked else "" 

1474 # UX-200: `spoken` names the glyph buttons for screen readers. 

1475 steps.button( 

1476 f"◀ {spoken('Previous trial')}", 

1477 key=f"{key_prefix}_prev_trial" if key_prefix else "prev_trial", 

1478 wrap=True, 

1479 on_click=_step_trial, 

1480 args=(-1,), 

1481 disabled=current_idx == 0 

1482 and (not linked or at_list_end(compare_snapshot, COMPARE_TRIAL_KEY, -1)), 

1483 help="Previous trial." + step_help, 

1484 ) 

1485 steps.button( 

1486 f"▶ {spoken('Next trial')}", 

1487 key=f"{key_prefix}_next_trial" if key_prefix else "next_trial", 

1488 wrap=True, 

1489 on_click=_step_trial, 

1490 args=(1,), 

1491 disabled=current_idx == n_trials - 1 

1492 and (not linked or at_list_end(compare_snapshot, COMPARE_TRIAL_KEY, 1)), 

1493 help="Next trial." + step_help, 

1494 ) 

1495 

1496 if not selected_trial_label: 

1497 return None, None, None 

1498 

1499 chosen = available_trials[ 

1500 available_trials[trial_field].astype(str) == selected_trial_label 

1501 ].iloc[0] 

1502 selected_text = str(chosen[text_field]) if text_field in chosen.index else None 

1503 return chosen["participant_id"], chosen["trial_id"], selected_text 

1504 

1505 

1506def select_trial( 

1507 combos: pd.DataFrame, 

1508 key_prefix: str = "", 

1509 picker_host=None, 

1510 *, 

1511 words: pd.DataFrame | None = None, 

1512 fixations: pd.DataFrame | None = None, 

1513 leading_renderer=None, 

1514 filter_renderer=None, 

1515 trailing_renderer=None, 

1516) -> tuple[str | None, str | None, str, str | None]: 

1517 """Pick a specific trial from the (already-narrowed) pool. 

1518 

1519 There are no Browse-by modes anymore, and no per-mapping variants either: the 

1520 pool is narrowed by the inline **Narrow by** Text / Participant multiselects + 

1521 the **More** filters (``controls.render_narrow_by`` / 

1522 ``render_trial_filters``), and this always picks one trial via the same 

1523 selectbox + slider + ◀ ▶ arrows — whatever the Trial ID mapping looks like. 

1524 

1525 BUG-23: a composite trial id (built from several mapped columns) used to get a 

1526 picker of its own — first one selector per mapped component, then a Participant 

1527 → Text cascade. Both made the *shape of the mapping* visible in the UI, and 

1528 neither offered the slider or the step buttons, so stepping through trials 

1529 worked on some datasets and not others. Participant and Text are what **Narrow 

1530 by** is for; the composite flag now only tells the chip strip to spell the 

1531 joined id out (``tabs._render_trial_condition_chips``). 

1532 

1533 ``picker_host`` is the container to render into (defaults to the current one); 

1534 the picker builds its own row of columns, so call it where columns are allowed. 

1535 

1536 ``words`` / ``fixations`` (optional) are the frames ``combos`` was built from; 

1537 passing them unlocks the UX-10 ⇅ sort popover's computed keys (fixation count, 

1538 reading time). Without them the sort still offers the column-based orderings. 

1539 

1540 Returns: 

1541 Tuple of (participant_id, trial_id, selection_mode, selected_text). 

1542 ``selection_mode`` is always ``"Trial"`` (kept for the comparison-options 

1543 builder, which still supports the other modes when called directly). 

1544 """ 

1545 if combos.empty: 

1546 st.warning( 

1547 "No trials match the filters. Clear one, or use ✕ Clear all filters." 

1548 ) 

1549 st.stop() 

1550 

1551 trial_field = ( 

1552 "unique_trial_id" if "unique_trial_id" in combos.columns else "trial_id" 

1553 ) 

1554 text_field = "unique_text_id" if "unique_text_id" in combos.columns else "text_id" 

1555 

1556 participant, trial, text = _select_trial_none_mode( 

1557 combos, 

1558 trial_field, 

1559 text_field, 

1560 key_prefix, 

1561 picker_host=picker_host, 

1562 # UX-64: the row's lead cell (the dataset picker) and its 🔎 filter 

1563 # popover are filled by the caller, which owns those widgets. 

1564 leading_renderer=leading_renderer, 

1565 filter_renderer=filter_renderer, 

1566 trailing_renderer=trailing_renderer, 

1567 words=words, 

1568 fixations=fixations, 

1569 ) 

1570 

1571 return participant, trial, "Trial", text 

1572 

1573 

1574# ----------------------------------------------------------------------------- 

1575# Statistics and metadata 

1576# ----------------------------------------------------------------------------- 

1577 

1578 

1579def compute_trial_stats( 

1580 trial_words: pd.DataFrame, trial_fixations: pd.DataFrame 

1581) -> dict[str, float]: 

1582 """Compute summary statistics for a single trial.""" 

1583 total_time = None 

1584 if "trial_dwell_time_ms" in trial_words.columns: 

1585 dwell_values = ( 

1586 pd.to_numeric(trial_words["trial_dwell_time_ms"], errors="coerce") 

1587 .dropna() 

1588 .unique() 

1589 ) 

1590 if len(dwell_values): 

1591 total_time = float(dwell_values[0]) 

1592 if total_time is None: 

1593 total_time = ( 

1594 float(trial_fixations["duration_ms"].sum()) 

1595 if not trial_fixations.empty 

1596 else 0.0 

1597 ) 

1598 return dict( 

1599 total_reading_time_ms=total_time, 

1600 total_reading_time_s=total_time / 1000.0, 

1601 word_count=len(trial_words), 

1602 fixation_count=len(trial_fixations), 

1603 ) 

1604 

1605 

1606def safe_summary(series: pd.Series) -> dict: 

1607 """Compute summary statistics for a series, handling empty data.""" 

1608 if series.empty: 

1609 nan_val = float("nan") 

1610 return dict(mean=nan_val, std=nan_val, min=nan_val, max=nan_val, median=nan_val) 

1611 return dict( 

1612 mean=float(series.mean()), 

1613 std=float(series.std(ddof=0)), 

1614 min=float(series.min()), 

1615 max=float(series.max()), 

1616 median=float(series.median()), 

1617 ) 

1618 

1619 

1620# ----------------------------------------------------------------------------- 

1621# Comparison helpers 

1622# ----------------------------------------------------------------------------- 

1623 

1624 

1625# Markers shown beside comparison-trial options. 

1626SAME_TEXT_MARKER = ( 

1627 "📄" # same stimulus text as the primary trial (★ reserved for favorites, UX-6) 

1628) 

1629SAME_PARTICIPANT_MARKER = "👤" # same participant as the primary trial 

1630 

1631 

1632def _allocate_label(label: str, qualified: str, used_labels: set[str]) -> str: 

1633 """``label`` if unused, else ``qualified``, else ``qualified (n)`` for the 

1634 first free ``n`` — always a label no earlier option holds (round 11: the 

1635 qualified form itself could already be taken, by a trial id that reads 

1636 like it, and the later option then overwrote the earlier one's identity).""" 

1637 candidate = label if label not in used_labels else qualified 

1638 n = 2 

1639 while candidate in used_labels: 

1640 candidate = f"{qualified} ({n})" 

1641 n += 1 

1642 used_labels.add(candidate) 

1643 return candidate 

1644 

1645 

1646def _compare_option_label( 

1647 participant_id: str, 

1648 trial_id: str, 

1649 markers: str, 

1650 used_labels: set[str], 

1651) -> str: 

1652 """Selectbox label for a comparison option: ``"<markers> <trial_id>"``. 

1653 

1654 De-duplicates on ``trial_id`` (two participants can share one) by appending 

1655 the participant in brackets — and a counter when even that is taken — so the 

1656 label stays a unique selectbox option / dict key in 

1657 ``tabs._render_compare_selector``.""" 

1658 trial_str = str(trial_id) if trial_id is not None else "" 

1659 prefix = f"{markers} " if markers else "" 

1660 return _allocate_label( 

1661 f"{prefix}{trial_str}", f"{prefix}{trial_str} [{participant_id}]", used_labels 

1662 ) 

1663 

1664 

1665def friendly_trial_label( 

1666 participant_id: str, 

1667 trial_id: str, 

1668 text_id: str | None, 

1669 existing_labels: set[str], 

1670 prefix: str = "", 

1671) -> str: 

1672 """Create a short, de-duplicated label for comparison dropdowns/legends.""" 

1673 trial_str = str(trial_id) if trial_id is not None else "" 

1674 text_str = str(text_id) if text_id is not None else "" 

1675 text_str = text_str.strip() 

1676 trial_contains_text = text_str and text_str.lower() in trial_str.lower() 

1677 

1678 if text_str: 

1679 base = f"{text_str} · {participant_id}" 

1680 if not trial_contains_text: 

1681 base = f"{base} (trial {trial_str})" if trial_str else base 

1682 elif trial_str != text_str: 

1683 # Surface any trial_id suffix beyond the text id (e.g. a 

1684 # repeat-reading "_r2" tag added during normalization). Without 

1685 # this the primary and compare titles look identical when a 

1686 # participant re-read the same text. 

1687 extra = trial_str 

1688 if extra.lower().startswith(text_str.lower()): 

1689 extra = extra[len(text_str) :].lstrip("_- ") 

1690 if extra: 

1691 base = f"{text_str} ({extra}) · {participant_id}" 

1692 else: 

1693 base = f"{trial_str} · {participant_id}" if trial_str else participant_id 

1694 

1695 return _allocate_label( 

1696 f"{prefix}{base}", f"{prefix}{base} [{trial_str or 'trial'}]", existing_labels 

1697 ) 

1698 

1699 

1700def build_comparison_options( 

1701 combos: pd.DataFrame, 

1702 selection_mode: str, 

1703 primary_participant: str, 

1704 primary_trial: str, 

1705 primary_text: str | None, 

1706 *, 

1707 cross_dataset: bool = False, 

1708 include_primary: bool = True, 

1709) -> list[tuple[str, str, str, str]]: 

1710 """Build a prioritized list of comparison-trial options. 

1711 

1712 Returns ``(participant_id, trial_id, label, markers)`` tuples, where 

1713 ``markers`` leads with the relation icons ``"📄"`` (same text) / ``"👤"`` (same 

1714 participant) and then the trial's annotation markers ``★`` (favorite) / ``🏷️`` 

1715 (tagged) / ``📝`` (noted) when present (UX-6), and ``label`` is 

1716 ``"<markers> <trial_id>"``. Ordered: same-text (📄) first, then same-participant 

1717 (👤), then the rest. A trial that is BOTH same-text and same-participant sorts 

1718 with the 📄 group (text-matches lead) and shows both markers. 

1719 

1720 ``cross_dataset`` (CMP-8 §5.1) says ``combos`` describes a *different* 

1721 dataset, and degrades the three id-based signals that would otherwise lie: 

1722 two corpora do not share readers, so 👤 never fires; a foreign trial's ★ / 

1723 🏷️ / 📝 are read from *its* dataset's annotations, never the active 

1724 dataset's, whose matching-looking ids name other trials (DATA-48 — they 

1725 were dropped altogether before annotations were per dataset); and the 

1726 primary trial is not in this pool, so a 

1727 coincidentally identical ``(participant, trial)`` is a real candidate rather 

1728 than the trial being compared. 📄 survives — a text id that matches across 

1729 corpora is exactly the pairing this feature exists for. 

1730 

1731 ``include_primary`` (CMP-22) keeps the selected trial itself in the pool, so 

1732 B's picker lists every trial A's does and the two position readouts agree. 

1733 It is only a *candidate*: the picker defaults B to the first trial that is 

1734 not A. Pass ``False`` to ask "is there anything else to compare with?" — 

1735 the question the Compare gate asks. 

1736 """ 

1737 text_field = "unique_text_id" if "unique_text_id" in combos.columns else "text_id" 

1738 uniq = combos.drop_duplicates(subset=["participant_id", "trial_id"]) 

1739 

1740 foreign_store = store_for_prefix("cmp") if cross_dataset else None 

1741 rows: list[dict] = [] 

1742 for row in uniq.itertuples(): 

1743 if ( 

1744 not include_primary 

1745 and not cross_dataset 

1746 and (row.participant_id, row.trial_id) 

1747 == (primary_participant, primary_trial) 

1748 ): 

1749 continue 

1750 text_id = getattr(row, text_field, "") 

1751 same_text = bool(primary_text and str(text_id) == str(primary_text)) 

1752 same_participant = not cross_dataset and bool( 

1753 str(row.participant_id) == str(primary_participant) 

1754 ) 

1755 markers = ( 

1756 (SAME_TEXT_MARKER if same_text else "") 

1757 + (SAME_PARTICIPANT_MARKER if same_participant else "") 

1758 + annotation_markers(row.participant_id, row.trial_id, store=foreign_store) 

1759 ) 

1760 rows.append( 

1761 { 

1762 "participant_id": row.participant_id, 

1763 "trial_id": row.trial_id, 

1764 "same_text": same_text, 

1765 "same_participant": same_participant, 

1766 "markers": markers, 

1767 } 

1768 ) 

1769 

1770 # ★ group first, then 👤 group, then the rest. same_text is the primary sort 

1771 # key so a both-★-👤 trial leads the ★ group. Stable sort preserves combos 

1772 # order within a group. 

1773 rows.sort( 

1774 key=lambda r: (0 if r["same_text"] else 1, 0 if r["same_participant"] else 1) 

1775 ) 

1776 

1777 used_labels: set[str] = set() 

1778 options: list[tuple[str, str, str, str]] = [] 

1779 for r in rows: 

1780 label = _compare_option_label( 

1781 r["participant_id"], r["trial_id"], r["markers"], used_labels 

1782 ) 

1783 options.append((r["participant_id"], r["trial_id"], label, r["markers"])) 

1784 return options 

1785 

1786 

1787# ----------------------------------------------------------------------------- 

1788# Cross-dataset comparison frames (CMP-8 · promoted from tabs.py for CMP-9) 

1789# ----------------------------------------------------------------------------- 

1790# These live here, not in tabs.py, because three surfaces now build a 

1791# cross-dataset comparison — the app, `api.compare_scanpaths` and 

1792# `cli.render --compare-*` — and the namespacing rule below is the one piece of 

1793# it that must not be re-derived per surface. `tests/test_compare_cross_dataset.py` 

1794# exists because getting it wrong renders a *plausible-looking wrong figure* 

1795# rather than an error. 

1796 

1797#: Separator between a dataset name and a participant id in a qualified id. 

1798COMPARE_DATASET_SEP = " · " 

1799 

1800 

1801def qualify_for_compare(frame: pd.DataFrame, dataset: str) -> pd.DataFrame: 

1802 """A copy of ``frame`` whose ``participant_id`` is namespaced by ``dataset``. 

1803 

1804 Two corpora can hold the same ``(participant_id, trial_id)``, and 

1805 `plots.make_comparison_figure` slices its frame by exactly that pair — so an 

1806 unqualified merge would silently render *the wrong scanpath*, or two. 

1807 

1808 Only ever applied to the single-trial frames that feed the comparison 

1809 builder. Nothing the annotations, the export slug, the deep link or Corpus 

1810 Analysis reads goes through here: those key on the real ids, and must. 

1811 """ 

1812 if frame.empty: 

1813 return frame 

1814 out = frame.copy() 

1815 out["dataset"] = dataset 

1816 out["participant_id"] = ( 

1817 dataset + COMPARE_DATASET_SEP + out["participant_id"].astype(str) 

1818 ) 

1819 return out 

1820 

1821 

1822def separate_self_compare(frame: pd.DataFrame, participant: str) -> pd.DataFrame: 

1823 """A copy of B's single-trial ``frame`` renamed apart from A's (CMP-22). 

1824 

1825 B may now be A's own trial. `plots.make_comparison_figure` slices its merged 

1826 frame by ``(participant_id, trial_id)``, so two copies of one trial would hand 

1827 *each* side both copies (and a duplicated index the word-line clustering 

1828 rejects). Giving B's copy `self_compare_participant`'s id keeps the halves 

1829 apart — the same trick `qualify_for_compare` plays across corpora, and just 

1830 as figure-only: labels, lookups, exports and links keep the real id. 

1831 """ 

1832 if frame.empty: 

1833 return frame 

1834 return frame.assign(participant_id=self_compare_participant(participant)) 

1835 

1836 

1837def self_compare_participant(participant: str) -> str: 

1838 """The id `separate_self_compare` gives B's copy of ``participant``.""" 

1839 return f"{participant}{COMPARE_DATASET_SEP}B" 

1840 

1841 

1842def qualified_participant(dataset: str, participant: str) -> str: 

1843 """The id `qualify_for_compare` gives ``participant`` inside ``dataset``.""" 

1844 return f"{dataset}{COMPARE_DATASET_SEP}{participant}" 

1845 

1846 

1847def unqualify_for_export(frame: pd.DataFrame, participant: str) -> pd.DataFrame: 

1848 """Undo `qualify_for_compare`'s rename, restoring the corpus' own id. 

1849 

1850 The namespace exists so `make_comparison_figure` can slice two colliding 

1851 ``(participant, trial)`` pairs apart. An exported table must carry the id the 

1852 corpus actually uses, or it won't join back to anything (CMP-8 §6). The 

1853 stamped ``dataset`` column is kept — that is what disambiguates the rows. 

1854 """ 

1855 if frame is None or frame.empty or "dataset" not in frame.columns: 

1856 return frame 

1857 out = frame.copy() 

1858 out["participant_id"] = str(participant) 

1859 return out 

1860 

1861 

1862def align_compare_columns( 

1863 a: pd.DataFrame, b: pd.DataFrame 

1864) -> tuple[pd.DataFrame, pd.DataFrame, frozenset]: 

1865 """Reindex two frames onto their column **union**, and report shared numerics. 

1866 

1867 A bare ``pd.concat`` of frames with disjoint columns warns and churns dtypes 

1868 (int columns become float once the other frame's rows fill in as NaN), which 

1869 matters here because two corpora rarely ship the same measure set. Aligning 

1870 first keeps the concat quiet and the dtypes stable. 

1871 

1872 The third element is the columns both frames carry *as the same kind* — 

1873 numeric in both, or categorical in both: what a cross-dataset figure may 

1874 legitimately colour by (CMP-8 §5.4; categorical since Compare colours by a 

1875 category too). A column present in only one corpus would colour one panel 

1876 and blank the other, and one numeric on one side only would be a scale on 

1877 one panel and a palette on the other. 

1878 """ 

1879 union = list(dict.fromkeys([*a.columns, *b.columns])) 

1880 a_aligned = a.reindex(columns=union) if list(a.columns) != union else a 

1881 b_aligned = b.reindex(columns=union) if list(b.columns) != union else b 

1882 shared = frozenset( 

1883 col 

1884 for col in set(a.columns) & set(b.columns) 

1885 if pd.api.types.is_numeric_dtype(a[col]) 

1886 == pd.api.types.is_numeric_dtype(b[col]) 

1887 ) 

1888 return a_aligned, b_aligned, shared