Coverage for scanpath_studio/export.py: 94%

1024 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Configurable bulk export of figures and tabular data for filtered trials. 

2 

3This module powers the "Bulk export" button. Users pick which artifacts they 

4want per trial (PNG, SVG, JSON plot config, fixations CSV/Parquet, per-word 

5measures CSV/Parquet), or each table once with every trial stacked in it 

6(EXP-23's *Combine all trials into one file*). Everything is packaged into a 

7single zip archive with a clean folder structure: 

8 

9 bulk_export_<timestamp>.zip 

10 ├─ per_trial/ 

11 │ ├─ <participant>__<trial>/ 

12 │ │ ├─ figure.png 

13 │ │ ├─ figure.svg 

14 │ │ ├─ layers/ (VIZ-5, optional) 

15 │ │ │ ├─ word_boxes.svg 

16 │ │ │ ├─ fixations.svg 

17 │ │ │ ├─ saccades.svg 

18 │ │ │ └─ … (one per visible layer) 

19 │ │ ├─ plot_config.json 

20 │ │ ├─ fixations.csv (and/or .parquet) 

21 │ │ └─ measures.csv (and/or .parquet) 

22 │ ├─ ... 

23 ├─ aggregate/ (combined tables, instead of the per-trial ones) 

24 │ ├─ all_fixations.csv (and/or .parquet) 

25 │ └─ all_measures.csv (and/or .parquet) 

26 └─ annotations.json (UX-179, optional) 

27""" 

28 

29from __future__ import annotations 

30 

31import io 

32import json 

33import re 

34import zipfile 

35from collections.abc import Mapping 

36from contextlib import contextmanager, nullcontext 

37from dataclasses import dataclass, field 

38from datetime import UTC, datetime 

39from pathlib import PurePosixPath 

40from time import perf_counter 

41 

42import pandas as pd 

43 

44from .aggregation import reader_summary_table, trial_summary_table 

45from .column_names import ( 

46 ColumnNames, 

47 across_tables, 

48 as_written, 

49 columns_manifest, 

50 dictionary_lines, 

51 written_columns, 

52) 

53from .constants import ( 

54 CITATION, 

55 DEFAULT_FIXATION_COLOR, 

56 DEFAULT_FIXATION_SYMBOL, 

57 DEFAULT_LINE_SPACING, 

58 DEFAULT_PALETTE, 

59 DEMO_CHOICE, 

60 ICONS, 

61 PLOTLY_CONFIG, 

62 SACCADE_CLASS_ORDER, 

63 UNIFORM_COLOR_FIELD, 

64 computed_measures_enabled, 

65 drift_correction_enabled, 

66) 

67from .data import brought_reading_measures, shareable_frame 

68from .export_status import ExportStage, StatusCallback, emit_status 

69from .fields import panel_field 

70from .measures import assign_fixations_to_words, enrich_fixations 

71from .multipart import ( 

72 SCREEN_ID, 

73 SCREEN_INDEX, 

74 extract_part, 

75 part_catalog, 

76 screen_canvas_size, 

77) 

78from .plots import ( 

79 SCANPATH_LAYER_ORDER, 

80 STATIC_FIGURE_OPTIONS, 

81 FigureSettings, 

82 _plotly_literal, 

83 make_scanpath_figure, 

84 normalize_legend_layout, 

85 split_scanpath_layers, 

86) 

87from .preprocessing import ( 

88 character_grid, 

89 cleaning_report, 

90 saccade_table, 

91 sentence_measures, 

92) 

93from .progress import report as report_progress 

94from .utils import extract_trial 

95 

96 

97def _local_stamp() -> str: 

98 """Now in this computer's local time, with its UTC offset said (#374, F28): 

99 ``2026-10-06 21:21:01 (UTC+03:00)``.""" 

100 now = datetime.now().astimezone() 

101 offset = now.strftime("%z") 

102 return f"{now:%Y-%m-%d %H:%M:%S} (UTC{offset[:3]}:{offset[3:]})" 

103 

104 

105# --- #374 F28 · a raster figure's print size --------------------------------- 

106#: The units a print width is given in → millimetres per unit. 

107PRINT_UNITS = {"mm": 1.0, "in": 25.4} 

108#: The resolution a print width is drawn at when none is named. 

109DEFAULT_PRINT_DPI = 300 

110#: #374 F28 — the print width and dpi every surface accepts: Export → Current 

111#: figure's boxes, a link or settings file (clamped), `render` and `save_figure` 

112#: (refused outside them). 

113PRINT_WIDTH_BOUNDS = (1.0, 2000.0) 

114PRINT_DPI_BOUNDS = (50, 2400) 

115#: What the app's Export → Current figure PNG is drawn at without a print width 

116#: (`tabs._PNG_EXPORT_SCALE`): three pixels per figure pixel. 

117SCREEN_PNG_SCALE = 3 

118 

119 

120def print_width_px(width: float, unit: str = "mm", dpi: int = DEFAULT_PRINT_DPI) -> int: 

121 """The pixel width of a raster figure ``width`` ``unit`` wide at ``dpi``: 

122 180 mm at 600 dpi is 4,252 px.""" 

123 if unit not in PRINT_UNITS: 

124 raise ValueError(f"Unknown width unit {unit!r}; use 'mm' or 'in'.") 

125 if not width or width <= 0 or not dpi or dpi <= 0: 

126 raise ValueError("A print width and its dpi must both be positive.") 

127 low, high = PRINT_WIDTH_BOUNDS 

128 if not low <= float(width) <= high: 

129 raise ValueError(f"A print width must be {low:g}–{high:g} {unit}.") 

130 low, high = PRINT_DPI_BOUNDS 

131 if not low <= float(dpi) <= high: 

132 raise ValueError(f"A print dpi must be {low}–{high}.") 

133 return max(1, round(float(width) * PRINT_UNITS[unit] / 25.4 * float(dpi))) 

134 

135 

136def print_scale( 

137 figure_width: int, width: float, unit: str = "mm", dpi: int = DEFAULT_PRINT_DPI 

138) -> float: 

139 """The ``scale`` that draws a ``figure_width``-px figure ``width`` ``unit`` 

140 wide at ``dpi`` (the height follows the figure's own aspect).""" 

141 return print_width_px(width, unit, dpi) / float(figure_width) 

142 

143 

144def png_save_kwargs( 

145 width: float | None, unit: str = "mm", dpi: int | None = None 

146) -> dict: 

147 """The `api.save_figure` keywords that write the PNG Export → *Current 

148 figure* writes: the print width at its dpi, else the screen size at 

149 `SCREEN_PNG_SCALE` — so Share → Code writes the same pixel size.""" 

150 if not width: 

151 return {"scale": SCREEN_PNG_SCALE} 

152 return {f"width_{unit}": float(width), "dpi": int(dpi or DEFAULT_PRINT_DPI)} 

153 

154 

155def set_png_dpi(path, dpi: int) -> None: 

156 """Stamp ``dpi`` into a PNG's header (``pHYs``), so a layout program 

157 places it at its print size. The pixels are untouched.""" 

158 from PIL import Image 

159 

160 with Image.open(path) as image: 

161 image.load() 

162 image.save(path, format="PNG", dpi=(dpi, dpi)) 

163 

164 

165# --- EXP-1 · customizable export paths --------------------------------------- 

166# A zip of 200 trials landed with names the tool chose, which is rarely how a 

167# user organizes figures for a paper. The path of every artifact is now a 

168# *pattern* over the trial's own fields, so a batch can be dropped straight into 

169# an existing folder structure. The default reproduces the historical layout 

170# exactly, so nothing changes unless the pattern is edited. 

171DEFAULT_PATH_PATTERN = "per_trial/{participant_id}__{trial_id}/{artifact}.{ext}" 

172# EXP-2 · a figure pulled into a paper or a slide loses its provenance the moment 

173# it leaves the app, so it can carry its own. Same substitution as the path 

174# patterns; empty means "no title / no caption". 

175DEFAULT_TITLE_PATTERN = "{participant_id} · {trial_id}" 

176DEFAULT_CAPTION_PATTERN = "{text_id} · {n_fixations} fixations · {settings}" 

177 

178 

179@dataclass 

180class ExportOptions: 

181 """User-chosen export artifacts. 

182 

183 Defaults: figures on (PNG + SVG), tabular data off. The scope fields 

184 narrow the set of trials when not "all". 

185 """ 

186 

187 include_png: bool = True 

188 include_svg: bool = True 

189 include_pdf: bool = False 

190 # HTML is a browser-free figure format (fig.to_html — no Kaleido/Chrome) and 

191 # stays interactive; handled specially in the export loop. 

192 include_html: bool = False 

193 include_plot_config: bool = True 

194 # UX-179: the exported trials' annotations (favorites, tags, notes) as one 

195 # `annotations.json` at the bundle root — the file 🗂️ Data → Annotations 

196 # exports and imports. Off by default: notes can be personal. 

197 include_annotations: bool = False 

198 include_fixations: bool = False 

199 # VIZ-45: each trial's raw (sample-level) gaze as its own table, written as 

200 # recorded — for a raw-gaze-only dataset it is the only recording there is. 

201 include_raw_gaze: bool = False 

202 # AN-32 / EXP-23: the word table with the reading measures the dataset 

203 # *brought*. Export computes none, as the Corpus Analysis page doesn't. 

204 include_measures: bool = False 

205 include_analysis_family: bool = False 

206 # EXP-23: write each chosen table once, every exported trial stacked, as 

207 # `aggregate/all_<table>` — instead of one file per trial. Replaced the 

208 # "Mega-table", which stacked two of the tables beside the per-trial ones. 

209 combine_trials: bool = False 

210 # VIZ-5: also drop a per-layer breakdown of the figure (word boxes / fixations 

211 # / saccades / heatmap / labels / stimulus image) into `layers/` so each can be 

212 # restyled independently in Illustrator / Inkscape. Uses the selected vector / 

213 # raster formats (SVG when none was picked — the publication default). 

214 separable_layers: bool = False 

215 table_format: str = "csv" # "csv" | "parquet" | "both" 

216 png_scale: int = 2 

217 # EXP-1: where each artifact lands inside the zip. `{artifact}` / `{ext}` name 

218 # the file (figure / fixations / measures / plot_config / layer names); every 

219 # other placeholder comes from the trial. The default is the historical layout. 

220 path_pattern: str = DEFAULT_PATH_PATTERN 

221 # EXP-2: an optional title + caption rendered into the exported image and 

222 # recorded in the per-trial manifest. Empty = off (the historical behaviour). 

223 title_pattern: str = "" 

224 caption_pattern: str = "" 

225 # VIZ-36: what `{dataset_name}` substitutes to. The app fills it from the 

226 # dataset picker's label; a headless caller that says nothing gets "", so 

227 # the placeholder renders empty rather than erroring on a surface nobody is 

228 # watching. 

229 dataset_name: str = "" 

230 # DATA-20 milestone 10: which columns of the attached participant table go 

231 # into `metadata/participants.*`. `None` is every field (the default, and 

232 # what a headless caller that knows nothing about this gets); a tuple is 

233 # exactly those, always alongside the reader id; an **empty** tuple leaves 

234 # the table out of the bundle. Reader attributes are the most re-identifying 

235 # thing an export can carry, so the opt-out has to be per field rather than 

236 # all-or-nothing. 

237 metadata_fields: tuple[str, ...] | None = None 

238 # DATA-29: the same per-field opt-out for the attached *trial* table, 

239 # written to `metadata/trials.*`. Kept as its own field rather than 

240 # folded into `metadata_fields` because the two tables are attached and 

241 # cleared independently, and because what is safe to ship differs by 

242 # grain: a reader attribute re-identifies a person, a trial attribute 

243 # usually describes the material. 

244 trial_metadata_fields: tuple[str, ...] | None = None 

245 # And the same opt-out for the attached *text* table, written to 

246 # `metadata/texts.*` — the third grain, same reasoning again. 

247 text_metadata_fields: tuple[str, ...] | None = None 

248 # HTML figures embed the Plotly library (opens offline, ~4.8 MB more per 

249 # file) instead of loading it from cdn.plot.ly when opened. Off by default: 

250 # the bundle's *Self-contained HTML* choice, shown while HTML is picked. 

251 html_self_contained: bool = False 

252 # When True, export operates on the whole loaded dataset, ignoring the 

253 # trial-filter funnel; the caller supplies the unfiltered frames. 

254 export_unfiltered: bool = False 

255 scope: str = "all" # "all" | "trial" | "participant" | "text" 

256 scope_participant: str | None = None 

257 scope_trial: str | None = None 

258 scope_text: str | None = None 

259 # Which screens of a multipart trial go in, by `screen_id` — one id across 

260 # trials (OneStop's `Paragraph`, MultiplEYE's `page_1`). `None` is every 

261 # screen; a trial showing none of the chosen screens is left out. 

262 screens: tuple[str, ...] | None = None 

263 

264 def any_table(self) -> bool: 

265 return ( 

266 self.include_fixations 

267 or self.include_raw_gaze 

268 or self.include_measures 

269 or self.include_analysis_family 

270 ) 

271 

272 def table_formats(self) -> list[str]: 

273 if self.table_format == "both": 

274 return ["csv", "parquet"] 

275 return [self.table_format] 

276 

277 def figure_formats(self) -> list[str]: 

278 formats: list[str] = [] 

279 if self.include_png: 

280 formats.append("png") 

281 if self.include_svg: 

282 formats.append("svg") 

283 if self.include_pdf: 

284 formats.append("pdf") 

285 if self.include_html: 

286 formats.append("html") 

287 return formats 

288 

289 def raster_formats(self) -> list[str]: 

290 """Figure formats that need Kaleido/Chrome (everything but HTML).""" 

291 return [f for f in self.figure_formats() if f != "html"] 

292 

293 def layer_formats(self) -> list[str]: 

294 """Formats for the per-layer breakdown (VIZ-5) — the selected non-HTML 

295 figure formats, or SVG when none was picked (vectors suit Illustrator). 

296 Empty when separable layers are off.""" 

297 if not self.separable_layers: 

298 return [] 

299 return self.raster_formats() or ["svg"] 

300 

301 def needs_figure(self) -> bool: 

302 """Whether the export builds each trial's figure at all (combined figure 

303 formats, or the per-layer breakdown).""" 

304 return bool(self.figure_formats()) or self.separable_layers 

305 

306 def needs_kaleido(self) -> bool: 

307 """Whether any figure render goes through Kaleido/Chrome (combined raster 

308 formats, or per-layer non-HTML formats).""" 

309 return bool(self.raster_formats()) or bool(self.layer_formats()) 

310 

311 

312@dataclass 

313class ExportProgress: 

314 total_trials: int 

315 finished_trials: int = 0 

316 bytes_written: int = 0 

317 errors: list[str] = field(default_factory=list) 

318 # EXP-24: what the build produced, for the summary under it — every file in 

319 # the zip, and the figure files (combined figures and per-layer files) 

320 # written and failed, one per format per trial. 

321 files_written: int = 0 

322 figures_written: int = 0 

323 figures_failed: int = 0 

324 trials_skipped: int = 0 

325 

326 

327@dataclass(frozen=True) 

328class ExportSummary: 

329 """EXP-24: how a finished bundle reads — ``level`` is ``"success"``, 

330 ``"warning"`` (some of it failed) or ``"error"`` (figures were asked for and 

331 none was made), and ``expand_errors`` opens the error list for the last.""" 

332 

333 level: str 

334 message: str 

335 expand_errors: bool 

336 

337 

338#: Session key of the current figure's *Self-contained HTML* choice — the figure's 

339#: and the replay's HTML follow it; the bundle asks its own 

340#: (``<key_prefix>_html_self_contained``). 

341HTML_SELF_CONTAINED_KEY = "export_html_self_contained" 

342 

343 

344def html_plotlyjs(self_contained: bool) -> bool | str: 

345 """``to_html``'s ``include_plotlyjs`` for a downloaded HTML file: the 

346 library embedded (opens offline), or loaded from cdn.plot.ly when opened.""" 

347 return True if self_contained else "cdn" 

348 

349 

350def missing_browser_note(static_formats: bool) -> str: 

351 """EXP-24: the bundle's browser prerequisite, when it is unmet, else ``""``. 

352 

353 The bundle draws PNG, SVG and PDF through Kaleido on the server, which 

354 needs Chrome, Chromium or Edge there; the current figure's PNG and SVG are 

355 saved by the reader's own browser and need none — a distinction the panel 

356 used to keep to itself, so a bundle on a machine without one built a zip 

357 of failures. Asked only when ``static_formats`` are picked. 

358 """ 

359 if not static_formats: 

360 return "" 

361 from .animation_export import chrome_available 

362 

363 if chrome_available(): 

364 return "" 

365 return ( 

366 "PNG, SVG and PDF bundle figures need Chrome, Chromium or Edge on the " 

367 "computer running Scanpath Studio, and none was found, so they will fail. " 

368 "Pick **HTML**, which needs no browser — or install one. The current " 

369 "figure's PNG and SVG downloads are made by your own browser." 

370 ) 

371 

372 

373def summarize_export(progress: ExportProgress, size_bytes: int) -> ExportSummary: 

374 """What a built bundle holds and what failed, in one line (EXP-24). 

375 

376 A generic "Ready" over a zip whose figures all failed read as a success, 

377 with the failures one click away in an expander. The line counts what was 

378 made and what wasn't; a build whose requested figures all failed is an 

379 error, with its list open. The zip stays downloadable either way — the 

380 files it does hold are good. 

381 """ 

382 

383 def plural(n: int, word: str) -> str: 

384 return f"{n:,} {word}{'' if n == 1 else 's'}" 

385 

386 parts = [ 

387 f"{plural(progress.files_written, 'file')} · {size_bytes / 1_048_576:.1f} MB" 

388 ] 

389 asked = progress.figures_written + progress.figures_failed 

390 if asked: 

391 parts.append(f"{progress.figures_written:,} of {plural(asked, 'figure')} made") 

392 if progress.figures_failed: 

393 parts.append(f"{progress.figures_failed:,} failed") 

394 if progress.trials_skipped: 

395 parts.append(f"{plural(progress.trials_skipped, 'trial')} skipped (no data)") 

396 message = " · ".join(parts) 

397 if asked and not progress.figures_written: 

398 return ExportSummary( 

399 "error", 

400 f"No figures were made · {message}. The errors are listed below.", 

401 True, 

402 ) 

403 if progress.errors: 

404 return ExportSummary("warning", f"Partly built · {message}", False) 

405 return ExportSummary("success", f"Ready · {message}", False) 

406 

407 

408def _safe_id(text: str) -> str: 

409 return "".join(c if c.isalnum() or c in "-_." else "_" for c in str(text)) 

410 

411 

412# Placeholders every pattern gets on top of the trial's own columns. 

413_PATTERN_EXTRA_FIELDS = ( 

414 "artifact", 

415 "ext", 

416 "n_fixations", 

417 "n_words", 

418 "reading_time_s", 

419 "settings", 

420) 

421_PLACEHOLDER_RE = re.compile(r"\{([^{}]*)\}") 

422 

423 

424def _settings_summary(settings: dict) -> str: 

425 """A one-line description of the settings that produced the figure (EXP-2).""" 

426 layers = [ 

427 name 

428 for name, key in ( 

429 ("boxes", "show_words"), 

430 ("text", "show_word_labels"), 

431 ("fixations", "show_fixations"), 

432 ("saccades", "show_saccades"), 

433 ("heatmap", "show_heatmap"), 

434 ) 

435 if settings.get(key) 

436 ] 

437 parts = [f"layers: {', '.join(layers) or 'none'}"] 

438 color_by = settings.get("color_by") 

439 if color_by and color_by != UNIFORM_COLOR_FIELD: 

440 parts.append(f"color by {color_by}") 

441 palette = settings.get("palette", DEFAULT_PALETTE) 

442 if palette and palette != DEFAULT_PALETTE: 

443 parts.append(f"{palette} palette") 

444 return " · ".join(parts) 

445 

446 

447#: VIZ-36 — the fields that can hold *two* values at once, because an overlay 

448#: draws two readings into one frame. Each gains an ``_a`` / ``_b`` variant. 

449PAIRED_PATTERN_FIELDS = ("dataset_name", "participant_id", "trial_id", "text_id") 

450 

451#: EXP-22 — the tables a pattern can name a field of, as ``{table.field}``, and 

452#: the heading each gets in the *Available fields* list. The metadata tables 

453#: plus the two data tables' saved fields; a qualified name is what tells two 

454#: tables' ``font_size`` apart, and what keeps them out of the plain list. 

455TABLE_PATTERN_LABELS = { 

456 "participants": "Participants table", 

457 "trials": "Trials table", 

458 "texts": "Texts table", 

459 "fixations": "Fixations table", 

460 "words": "Words table", 

461} 

462 

463#: Columns a data table always carries or the app derives — the trial's own 

464#: identity, geometry and timing, which the plain fields already cover or which 

465#: are not one value per trial. What is left is what the user kept. 

466_CORE_TABLE_COLUMNS = frozenset( 

467 { 

468 "participant_id", 

469 "trial_id", 

470 "text_id", 

471 "paragraph_id", 

472 "unique_trial_id", 

473 "unique_text_id", 

474 "unique_paragraph_id", 

475 "word_id", 

476 "text", 

477 "line_idx", 

478 "x", 

479 "y", 

480 "width", 

481 "height", 

482 "screen_id", 

483 "screen_index", 

484 "canvas_width", 

485 "canvas_height", 

486 "screen_timestamp_ms", 

487 "screen_fixation_id", 

488 "duration_ms", 

489 "timestamp_ms", 

490 "fixation_id", 

491 "order_in_trial", 

492 "pass_index", 

493 "saccade_type", 

494 "saccade_amplitude", 

495 "eye", 

496 "source_file", 

497 "TRIAL_INDEX", 

498 "trial_index", 

499 } 

500) 

501 

502 

503def _saved_table_fields(frame: pd.DataFrame | None) -> dict: 

504 """A data table's saved fields that hold one value for the whole trial. 

505 

506 A field that varies within the trial (a word's surprisal, a fixation's 

507 pupil size) has no single value to put in a title, so it is not offered.""" 

508 out: dict = {} 

509 if frame is None or getattr(frame, "empty", True): 

510 return out 

511 for column in frame.columns: 

512 name = str(column) 

513 if name.startswith("_") or name in _CORE_TABLE_COLUMNS: 

514 continue 

515 values = frame[column].dropna() 

516 if values.empty: 

517 continue 

518 try: 

519 distinct = values.unique() 

520 except TypeError: # unhashable cells (lists, dicts) 

521 continue 

522 if len(distinct) != 1 or isinstance(distinct[0], (list, dict, set, tuple)): 

523 continue 

524 out[name] = distinct[0] 

525 return out 

526 

527 

528def table_pattern_fields( 

529 trial_words: pd.DataFrame | None, 

530 trial_fixations: pd.DataFrame | None, 

531 metadata_rows: dict | None = None, 

532) -> dict[str, dict]: 

533 """``{table: {"table.field": value}}`` for one trial (EXP-22). 

534 

535 ``metadata_rows`` is this trial's row of each attached metadata table 

536 (``metadata.pattern_rows``); the fixations and AOI tables contribute their 

537 saved fields that are constant within the trial. Grouped by table so the 

538 *Available fields* list can head each group; :func:`pattern_fields` 

539 flattens it.""" 

540 tables: dict[str, dict] = {} 

541 for table, row in (metadata_rows or {}).items(): 

542 if row: 

543 tables[table] = {f"{table}.{name}": value for name, value in row.items()} 

544 for table, frame in (("fixations", trial_fixations), ("words", trial_words)): 

545 saved = _saved_table_fields(frame) 

546 if saved: 

547 tables[table] = {f"{table}.{name}": value for name, value in saved.items()} 

548 return tables 

549 

550 

551def pattern_fields( 

552 participant: str, 

553 trial: str, 

554 trial_words: pd.DataFrame, 

555 trial_fixations: pd.DataFrame, 

556 settings: dict, 

557 combo_row: dict | None = None, 

558 dataset_name: str = "", 

559 compare_row: dict | None = None, 

560 metadata_rows: dict | None = None, 

561 column_names: ColumnNames | None = None, 

562) -> dict: 

563 """Every value a filename / title / caption pattern can substitute. 

564 

565 DATA-66: with ``column_names`` (the dataset's map), every field its file 

566 named is also reachable under that name — ``{RECORDING_SESSION_LABEL}`` and 

567 ``{participant_id}`` both work, so a pattern can be written in either 

568 vocabulary. 

569 

570 The trial's own combo columns (participant, trial, text, conditions …) plus 

571 counts and the settings summary. Values are left raw here; path rendering 

572 sanitizes them, while titles and captions want them readable. 

573 

574 **VIZ-36 — ``dataset_name`` arrives as an argument, never read from here.** 

575 The app knows it as ``data_source_choice`` (since DATA-9 that key *is* the 

576 picker's label, and DATA-23's rename re-keys it), but this function is pure 

577 and also runs headless under ``api.save_figure_layers`` and ``cli render``, 

578 where there is no session at all. Each of the five callers supplies it. 

579 

580 ``compare_row`` is the *other* reading in an overlay — two scanpaths in one 

581 frame, so a single ``{dataset_name}`` is ambiguous exactly where a title 

582 most wants to name both. Every field in :data:`PAIRED_PATTERN_FIELDS` gains 

583 an ``_a`` / ``_b`` variant, which are defined **always** (``_b`` empty when 

584 there is no second reading) so that a pattern written in compare mode still 

585 validates and renders on a single-trial figure instead of erroring on a 

586 surface the author cannot see. 

587 

588 EXP-22: every attached metadata table's fields and each data table's saved 

589 fields join as ``{table.field}`` (:func:`table_pattern_fields`) — qualified, 

590 so a trial table's ``font_size`` and a recorded ``font_size`` are both 

591 reachable, and none of the plain names above changes. 

592 """ 

593 fields: dict = dict(combo_row or {}) 

594 for table in table_pattern_fields( 

595 trial_words, trial_fixations, metadata_rows 

596 ).values(): 

597 fields.update(table) 

598 fields.update( 

599 participant_id=participant, 

600 trial_id=trial, 

601 dataset_name=dataset_name, 

602 n_fixations=len(trial_fixations), 

603 n_words=len(trial_words), 

604 reading_time_s=round( 

605 float( 

606 pd.to_numeric(trial_fixations.get("duration_ms"), errors="coerce").sum() 

607 ) 

608 / 1000.0, 

609 1, 

610 ) 

611 if "duration_ms" in getattr(trial_fixations, "columns", []) 

612 else 0.0, 

613 settings=_settings_summary(settings), 

614 ) 

615 fields.setdefault("text_id", trial) 

616 for name in PAIRED_PATTERN_FIELDS: 

617 fields[f"{name}_a"] = fields.get(name, "") 

618 fields[f"{name}_b"] = (compare_row or {}).get(name, "") 

619 if column_names is not None: 

620 # The counts and the settings summary are the app's, whatever a column 

621 # of the same canonical name was called in the file. 

622 own = {"n_fixations", "n_words", "reading_time_s", "settings", "dataset_name"} 

623 named = [field for field in fields if field not in own] 

624 for canonical, header in column_names.export_headers(named).items(): 

625 fields.setdefault(header, fields[canonical]) 

626 return fields 

627 

628 

629def pattern_error(pattern: str, fields: dict) -> str | None: 

630 """A human message naming any unknown placeholder, or ``None`` if valid. 

631 

632 Validated up front (and shown live in the UI) rather than at export time — 

633 discovering a typo after a 200-trial render is the worst place to find it. 

634 """ 

635 known = set(fields) | set(_PATTERN_EXTRA_FIELDS) 

636 unknown = [name for name in _PLACEHOLDER_RE.findall(pattern) if name not in known] 

637 if not unknown: 

638 return None 

639 return ( 

640 f"Unknown field{'s' if len(unknown) > 1 else ''}: " 

641 f"{', '.join('{' + u + '}' for u in unknown)}. " 

642 + ( 

643 f"Available: {', '.join('{' + k + '}' for k in sorted(known))}." 

644 if len(known) <= 25 

645 else f"{len(known)} fields are available; the app's Fields list shows them." 

646 ) 

647 ) 

648 

649 

650_DRIVE_PREFIX = re.compile(r"^[A-Za-z]:") 

651 

652 

653def path_structure_error(pattern: str) -> str | None: 

654 """What is wrong with ``pattern``'s own text as a path inside the ZIP, or 

655 ``None`` (round 10). 

656 

657 The values put into ``{…}`` are sanitized one by one (:func:`render_pattern`), 

658 but the text around them is the pattern's: ``../{artifact}.{ext}`` wrote a 

659 member outside the archive's root, and an empty pattern one with no name. 

660 Every member must be a relative path of named folders ending in a file 

661 name, so this refuses an empty pattern, a leading ``/`` or drive, a 

662 backslash (a separator to some unzip tools), and an empty, ``.`` or ``..`` 

663 folder or file name. Checked before any figure is rendered, in the app and 

664 by :func:`bulk_export`. 

665 """ 

666 text = str(pattern or "") 

667 probe = _PLACEHOLDER_RE.sub("x", text) 

668 if not probe.strip(): 

669 return "The file path pattern is empty." 

670 if "\\" in probe: 

671 return ( 

672 "Use `/` between folders: a backslash is a separator to some unzip tools." 

673 ) 

674 if probe.startswith("/") or _DRIVE_PREFIX.match(probe): 

675 return "The file path must be relative to the ZIP: start it with a folder or file name." 

676 for part in probe.split("/"): 

677 if not part.strip(): 

678 return "The file path has an empty folder or file name (`//`, or a trailing `/`)." 

679 if set(part) <= {"."}: 

680 return f"`{part}` can't be a folder or file name in the ZIP." 

681 return None 

682 

683 

684def _path_component(text: str) -> str: 

685 """One path segment, sanitized. ``.`` / ``..`` collapse so nothing escapes.""" 

686 safe = _safe_id(text) 

687 return "_" if set(safe) <= {"."} else safe 

688 

689 

690def render_pattern( 

691 pattern: str, 

692 fields: dict, 

693 *, 

694 as_path: bool = False, 

695 multi_segment_fields: tuple = (), 

696) -> str: 

697 """Substitute ``fields`` into ``pattern``. 

698 

699 ``as_path`` sanitizes each substituted value, so a *data* value containing 

700 ``/`` or ``..`` becomes one flat segment and can't escape the folder the 

701 pattern describes; the pattern's own ``/`` stay real separators. 

702 ``multi_segment_fields`` names the tool-controlled fields allowed to expand 

703 into several segments (``artifact``, which carries ``layers/<name>``) — each 

704 of their segments is still sanitized individually. A missing or null value 

705 becomes ``na`` rather than failing the whole export. 

706 """ 

707 

708 def _sub(match: re.Match) -> str: 

709 name = match.group(1) 

710 value = fields.get(name, "") 

711 if value is None or (isinstance(value, float) and pd.isna(value)): 

712 value = "na" 

713 text = str(value) 

714 if not as_path: 

715 return text 

716 if name in multi_segment_fields: 

717 return "/".join(_path_component(part) for part in text.split("/")) 

718 return _path_component(text) 

719 

720 return _PLACEHOLDER_RE.sub(_sub, pattern) 

721 

722 

723def resolve_export_path( 

724 pattern: str, fields: dict, *, artifact: str, ext: str, used: set 

725) -> str: 

726 """The zip path for one artifact, de-duplicated against ``used``. 

727 

728 Two trials can render to the same path (a pattern that omits the trial id, 

729 say). Writing both would put two entries at one name in the zip and silently 

730 lose one, so the second gets a ``-2`` suffix instead. ``used`` is mutated. 

731 """ 

732 path = render_pattern( 

733 pattern, 

734 {**fields, "artifact": artifact, "ext": ext}, 

735 as_path=True, 

736 multi_segment_fields=("artifact",), 

737 ).lstrip("/") 

738 if path not in used: 

739 used.add(path) 

740 return path 

741 stem, dot, suffix = path.rpartition(".") 

742 base, tail = (stem, f"{dot}{suffix}") if dot else (path, "") 

743 n = 2 

744 while f"{base}-{n}{tail}" in used: 

745 n += 1 

746 path = f"{base}-{n}{tail}" 

747 used.add(path) 

748 return path 

749 

750 

751# --- EXP-2 · titles and captions on the exported figure ----------------------- 

752# Sized bands rather than Plotly's automatic title spacing: the scanpath figure 

753# is equal-aspect (`scaleanchor`), so anything that eats into the plot area 

754# shrinks the WHOLE plot — and the true-to-scale word labels, computed for the 

755# un-shrunk size, then no longer match their boxes. Same constraint the animation 

756# transport controls hit; same fix: grow the figure by exactly what the band 

757# takes, so the plot region is untouched. 

758_TITLE_BAND_PX = 46 

759_CAPTION_LINE_PX = 22 

760_CAPTION_PAD_PX = 12 

761 

762 

763def annotate_figure(fig, *, title: str = "", caption: str = "") -> None: 

764 """Stamp ``title`` / ``caption`` onto ``fig`` in place, without shrinking it. 

765 

766 The figure grows by the height of each band and its margin grows to match, so 

767 the plotting area — and therefore the true-to-scale text — is byte-identical 

768 to the untitled figure. Both are drawn as written: ``<b>`` in a title shows 

769 as ``<b>``, and only a real newline starts a new caption line. 

770 """ 

771 if not title and not caption: 

772 return 

773 margin = fig.layout.margin 

774 height = fig.layout.height 

775 if title: 

776 fig.layout.margin.t = (margin.t or 0) + _TITLE_BAND_PX 

777 if height: 

778 height += _TITLE_BAND_PX 

779 fig.layout.height = height 

780 fig.update_layout( 

781 title=dict( 

782 text=_plotly_literal(title), 

783 x=0.5, 

784 xanchor="center", 

785 y=1.0, 

786 yanchor="top", 

787 pad=dict(t=14), 

788 font=dict(size=20), 

789 ) 

790 ) 

791 if caption: 

792 original_bottom = fig.layout.margin.b or 0 

793 band = _CAPTION_LINE_PX * (caption.count("\n") + 1) + _CAPTION_PAD_PX 

794 fig.layout.margin.b = original_bottom + band 

795 if height: 

796 fig.layout.height = height + band 

797 # Anchored to the plot's bottom edge and pushed into the space just 

798 # added, so it never overlaps whatever already lived in that margin. 

799 fig.add_annotation( 

800 text=_plotly_literal(caption).replace("\n", "<br>"), 

801 xref="paper", 

802 yref="paper", 

803 x=0, 

804 y=0, 

805 xanchor="left", 

806 yanchor="top", 

807 yshift=-(original_bottom + _CAPTION_PAD_PX // 2), 

808 showarrow=False, 

809 align="left", 

810 font=dict(size=13, color="#555555"), 

811 ) 

812 

813 

814# DATA-16 (security audit S4). Columns that hold a filesystem path from the 

815# machine the app ran on. `image_path` is a `passthrough` meta field on both 

816# schemas, so it survives normalization and rides into the exported fixation 

817# tables — and a fixations CSV is exactly the file that gets attached to a paper, 

818# posted to OSF, or mailed to a collaborator. `/Users/<name>/` discloses the OS 

819# account; the rest discloses the directory layout, including where a MultiplEYE 

820# corpus lives. The basename still identifies the stimulus, which is all the 

821# column is used for downstream. 

822# 

823# `source_file` is deliberately NOT here: it is an identity label, not a path. 

824# `data.source_labels` stores the file's stem, qualified only by the trailing 

825# folders that tell two same-named files apart — never the folders they share, 

826# so an absolute path's `/Users/<name>/…` prefix does not reach it. 

827_PATH_COLUMNS = ("image_path",) 

828 

829 

830def strip_local_paths(df: pd.DataFrame) -> pd.DataFrame: 

831 """Reduce path-bearing columns to their basename (S4). 

832 

833 Returns ``df`` unchanged (the same object) when it carries none of them, so 

834 the common case costs one membership test and no copy. 

835 """ 

836 present = [c for c in _PATH_COLUMNS if c in df.columns] 

837 if not present: 

838 return df 

839 out = df.copy() 

840 for column in present: 

841 values = out[column] 

842 # `na_action="ignore"` is load-bearing: pandas evaluates the `other` 

843 # argument of `.where` eagerly, over every row including the missing 

844 # ones, and since pandas 3 `astype(str)` leaves NaN as a float instead 

845 # of stringifying it to "nan" — so the lambda would see a float. 

846 out[column] = values.where( 

847 values.isna(), 

848 values.astype(str).map( 

849 lambda text: PurePosixPath(text.replace("\\", "/")).name, 

850 na_action="ignore", 

851 ), 

852 ) 

853 return out 

854 

855 

856#: DATA-66: the artifacts that *are* one of the dataset's tables (rows of it, 

857#: perhaps with computed columns added), and which one — so a fixation table's 

858#: `x` is named as the fixation file named it and a word table's `x` as the AOI 

859#: file did. Every other artifact is derived (a saccade table, a summary, a 

860#: character grid) and reuses canonical names for values of its own, so only its 

861#: id columns take the file's names (`ColumnNames.identity`). 

862_ARTIFACT_TABLE = { 

863 "fixations": "fixations", 

864 "raw_gaze": "raw_gaze", 

865 "measures": "words", 

866 "word_measures": "words", 

867} 

868 

869 

870def _write_table( 

871 zf: zipfile.ZipFile, 

872 path: str, 

873 df: pd.DataFrame, 

874 fmt: str, 

875 names: ColumnNames | None = None, 

876 hidden: set[str] | None = None, 

877) -> int: 

878 # DATA-49: the pipeline's bookkeeping columns stay out of what is shared. 

879 df = strip_local_paths(shareable_frame(df)) 

880 # DATA-66: and the columns the user's file named go out under those names. 

881 df = as_written(df, names, hidden) 

882 if fmt == "parquet": 

883 buf = io.BytesIO() 

884 df.to_parquet(buf, index=False) 

885 data = buf.getvalue() 

886 else: 

887 data = df.to_csv(index=False).encode("utf-8") 

888 zf.writestr(path, data) 

889 return len(data) 

890 

891 

892@contextmanager 

893def _figure_renderer(enabled: bool): 

894 """Yield ``render(fig, fmt, width, height, scale) -> bytes``. 

895 

896 When ``enabled`` and Kaleido starts, every trial's figure is rasterized 

897 through one persistent Kaleido browser (``calc_fig_sync``) instead of 

898 cold-starting a fresh Chrome on each ``fig.to_image`` call — the cold start 

899 is the "Resorting to unclean kill browser." log noise and ~seconds-per-trial 

900 latency. Falls back to per-call ``to_image`` if the warm server can't start 

901 (or no figures were requested), so behavior is unchanged when Kaleido/Chrome 

902 is unavailable — the per-trial failure is still surfaced as an export error. 

903 

904 ``enabled`` also holds ``animation_export.KALEIDO_LOCK`` until the server has 

905 stopped, since that server is one per process (UX-150). 

906 """ 

907 from .animation_export import KALEIDO_LOCK 

908 

909 with KALEIDO_LOCK if enabled else nullcontext(): 

910 server = None 

911 if enabled: 

912 try: 

913 import kaleido 

914 

915 from .animation_export import chromium_browser_path 

916 

917 browser_path = chromium_browser_path() 

918 if browser_path is not None: 

919 kaleido.start_sync_server(path=browser_path, silence_warnings=True) 

920 server = kaleido 

921 except Exception: 

922 server = None 

923 

924 def render(fig, fmt: str, width: int, height: int, scale: int) -> bytes: 

925 if server is not None: 

926 data = server.calc_fig_sync( 

927 fig, 

928 opts={ 

929 "format": fmt, 

930 "width": int(width), 

931 "height": int(height), 

932 "scale": scale, 

933 }, 

934 ) 

935 return bytes(data) 

936 return fig.to_image( 

937 format=fmt, width=int(width), height=int(height), scale=scale 

938 ) 

939 

940 try: 

941 yield render 

942 finally: 

943 if server is not None: 

944 try: 

945 server.stop_sync_server(silence_warnings=True) 

946 except Exception: # pragma: no cover - best-effort teardown 

947 pass 

948 

949 

950def render_static_figure_bytes( 

951 fig, 

952 *, 

953 fmt: str, 

954 width: int, 

955 height: int, 

956 scale: float, 

957 status_callback: StatusCallback | None = None, 

958) -> bytes: 

959 """Render one static figure with observable indeterminate job stages.""" 

960 from .animation_export import CHROME_INSTALL_HINT, chrome_available 

961 

962 started = perf_counter() 

963 emit_status( 

964 status_callback, 

965 ExportStage.PREPARING, 

966 "Preparing figure and checking export settings…", 

967 started_at=started, 

968 ) 

969 try: 

970 if not chrome_available(): 

971 raise RuntimeError(CHROME_INSTALL_HINT) 

972 emit_status( 

973 status_callback, 

974 ExportStage.STARTING_RENDERER, 

975 "Starting the Chrome/Kaleido renderer (cold starts can take a few seconds)…", 

976 started_at=started, 

977 ) 

978 with _figure_renderer(True) as render: 

979 emit_status( 

980 status_callback, 

981 ExportStage.RASTERIZING, 

982 f"Rendering {fmt.upper()}…", 

983 started_at=started, 

984 ) 

985 data = render(fig, fmt.lower(), int(width), int(height), float(scale)) 

986 emit_status( 

987 status_callback, 

988 ExportStage.FINALIZING, 

989 "Finishing the file…", 

990 started_at=started, 

991 ) 

992 result = bytes(data) 

993 emit_status( 

994 status_callback, 

995 ExportStage.READY, 

996 "Ready to download.", 

997 started_at=started, 

998 ) 

999 return result 

1000 except Exception as exc: 

1001 emit_status( 

1002 status_callback, 

1003 ExportStage.ERROR, 

1004 "Export failed.", 

1005 started_at=started, 

1006 error=str(exc), 

1007 ) 

1008 raise 

1009 

1010 

1011def _drift_corrected_for_figure( 

1012 fix: pd.DataFrame, words: pd.DataFrame, settings: dict 

1013) -> tuple[pd.DataFrame, tuple | None]: 

1014 """PRE-3 drift correction for one exported figure (EXP-4 / VIZ-24). 

1015 

1016 Returns ``(figure_fixations, connector_y)``. When no algorithm is selected 

1017 (``align_algorithm`` absent / ``"Off"`` — the default) or there is nothing to 

1018 correct, hands back the very same frame object and ``None`` — a true no-op, 

1019 mirroring ``tabs._drift_corrected``. Otherwise the returned frame has each 

1020 fixation's ``y`` snapped to its assigned text line, and ``connector_y`` 

1021 carries the *original* y values when ``align_connectors`` is on (the faint 

1022 original→corrected connector layer). 

1023 

1024 Deliberate asymmetry: this feeds the **figure only** — the exported tables 

1025 (fixations, measures, combined tables) stay uncorrected, because the correction 

1026 is a view on the data, not a rewrite of it.""" 

1027 algorithm = settings.get("align_algorithm") 

1028 if ( 

1029 not algorithm 

1030 or str(algorithm) == "Off" 

1031 or fix is None 

1032 or fix.empty 

1033 or words is None 

1034 or words.empty 

1035 ): 

1036 return fix, None 

1037 from .alignment import correct # local: pulls in scipy only when used 

1038 

1039 corrected, _ = correct(fix, words, method=str(algorithm).lower()) 

1040 connector_y = None 

1041 if settings.get("align_connectors") and "y" in fix.columns: 

1042 connector_y = tuple(pd.to_numeric(fix["y"], errors="coerce")) 

1043 return corrected, connector_y 

1044 

1045 

1046def _plot_config_dict( 

1047 participant: str, 

1048 trial: str, 

1049 canvas_width: int, 

1050 canvas_height: int, 

1051 x_field: str, 

1052 y_field: str, 

1053 settings: dict, 

1054 *, 

1055 screen_id: str | None = None, 

1056 drift_applied: bool = False, 

1057) -> dict: 

1058 selection = {"participant_id": participant, "trial_id": trial} 

1059 if screen_id not in (None, ""): 

1060 selection[SCREEN_ID] = str(screen_id) 

1061 return { 

1062 "selection": selection, 

1063 "canvas_px": {"width": int(canvas_width), "height": int(canvas_height)}, 

1064 "axes": { 

1065 "x_field": x_field, 

1066 "y_field": y_field, 

1067 "coordinate_grid": bool(settings.get("show_coordinate_grid", False)), 

1068 "coordinate_grid_auto": settings.get("coordinate_grid_spacing") is None, 

1069 "coordinate_grid_spacing": settings.get("coordinate_grid_spacing"), 

1070 }, 

1071 "layers": { 

1072 "words": settings.get("show_words"), 

1073 "word_labels": settings.get("show_word_labels"), 

1074 "fixations": settings.get("show_fixations"), 

1075 "order_labels": settings.get("show_order"), 

1076 "saccades": settings.get("show_saccades"), 

1077 "saccade_arrows": settings.get("show_saccade_arrows", False), 

1078 "heatmap": settings.get("show_heatmap"), 

1079 "raw_gaze": settings.get("show_raw_gaze"), 

1080 }, 

1081 "coloring": { 

1082 "color_by": settings.get("color_by"), 

1083 "heatmap_metric": settings.get("heatmap_metric"), 

1084 "heatmap_style": settings.get("heatmap_style", "Word boxes"), 

1085 "fixation_colorscale": settings.get("fixation_colorscale"), 

1086 "heatmap_colorscale": settings.get("heatmap_colorscale"), 

1087 # VIZ-18 palette · VIZ-17 flat colour · VIZ-15 shape — part of how the 

1088 # figure looked, so the manifest records them for reproduction. 

1089 "palette": settings.get("palette", DEFAULT_PALETTE), 

1090 "fixation_color": settings.get("fixation_color", DEFAULT_FIXATION_COLOR), 

1091 "fixation_symbol": settings.get("fixation_symbol", DEFAULT_FIXATION_SYMBOL), 

1092 "saccade_color_mode": settings.get("saccade_color_mode", "Uniform"), 

1093 # VIZ-31: which reading classes the exported figures actually drew — 

1094 # a regressions-only batch has to say so, or the files look like a 

1095 # dataset with almost no saccades in it. 

1096 "saccade_classes": list( 

1097 settings.get("saccade_classes") or SACCADE_CLASS_ORDER 

1098 ), 

1099 # EXP-4 / VIZ-24: which PRE-3 drift correction produced the exported 

1100 # figure ("Off" = none). The exported tables stay uncorrected — the 

1101 # manifest is where that split is recorded. Same keys as the 💾 Save 

1102 # & restore config (ENG-23). `color_by_line` records the EFFECTIVE 

1103 # value: a corrected figure is force-coloured by line, like the 

1104 # on-screen static path. 

1105 # PRE-21: omitted entirely while drift correction is gated off — 

1106 # recording `"drift_correction": "Off"` in every bundle would 

1107 # advertise a control the build doesn't have. 

1108 **( 

1109 { 

1110 "drift_correction": str( 

1111 settings.get("align_algorithm", "Off") or "Off" 

1112 ), 

1113 "drift_connectors": bool(settings.get("align_connectors", False)), 

1114 } 

1115 if drift_correction_enabled() 

1116 else {} 

1117 ), 

1118 "color_by_line": bool(settings.get("color_by_line", False)) 

1119 or drift_applied, 

1120 }, 

1121 "sizing": { 

1122 "marker_size_range": list(settings.get("marker_size_range", [])), 

1123 "marker_size_scale": settings.get("marker_size_scale"), 

1124 "marker_duration_range": list(settings.get("marker_duration_range") or []), 

1125 "duration_size_legend": bool(settings.get("duration_size_legend", True)), 

1126 "order_font_size": settings.get("order_font_size"), 

1127 }, 

1128 # Where each legend was placed (Figure & canvas → Legends); absent = Auto. 

1129 # Every legend, Auto included, as the settings file writes them: a 

1130 # partial section would leave a restoring session's moved legends. 

1131 "legends": normalize_legend_layout(settings.get("legend_layout")), 

1132 # True-to-scale reading text: records how the word labels were sized so 

1133 # the figure can be reproduced exactly (see plots._word_label_font_px). 

1134 "text": { 

1135 "scale_text_to_boxes": settings.get("scale_text_to_boxes", True), 

1136 "line_spacing": settings.get("line_spacing", DEFAULT_LINE_SPACING), 

1137 }, 

1138 # DATA-22 §7 surface 4: the recording setup + how each group is known, so 

1139 # an exported figure set records that (say) its monitor size was assumed. 

1140 # Sits beside `coloring.drift_correction`, which makes the same kind of 

1141 # "what produced these files" statement. Omitted when the source declared 

1142 # no setup — an absent key means unknown, which is the truth. 

1143 **({"experimental_setup": setup} if (setup := _setup_section()) else {}), 

1144 } 

1145 

1146 

1147def _setup_section() -> dict | None: 

1148 """The active source's `SetupSnapshot` as a dict, or ``None``. 

1149 

1150 Imported lazily and defensively: `bulk_export` also runs headlessly (the CLI 

1151 and `api.py`), where there is no session to read a snapshot from, and an 

1152 export must never fail because it could not describe its own geometry. 

1153 """ 

1154 try: 

1155 from scanpath_studio.app import active_setup_snapshot 

1156 

1157 snapshot = active_setup_snapshot() 

1158 except Exception: # pragma: no cover - headless / no session 

1159 return None 

1160 return snapshot.to_dict() if snapshot is not None else None 

1161 

1162 

1163def _render_scope_picker( 

1164 st, 

1165 combos: pd.DataFrame, 

1166 key_prefix: str, 

1167 combos_all: pd.DataFrame | None = None, 

1168 selected_participant: str | None = None, 

1169 selected_trial: str | None = None, 

1170) -> tuple[str, str | None, str | None, str | None, bool]: 

1171 """Render the scope radio + dependent selectors. 

1172 

1173 Returns ``(scope, pid, trial, text, export_unfiltered)``. The whole-dataset 

1174 choice lives inside the "Trials to include" radio (an extra "All" option that 

1175 ignores the trial filters) rather than as a separate checkbox. 

1176 """ 

1177 # Build the ordered radio: label -> (scope, export_unfiltered). Both "All" 

1178 # (the whole dataset, ignoring the trial filters) and "All filtered trials" 

1179 # (the current filter selection) are always offered — they coincide only 

1180 # when no filter is active. 

1181 options_map: dict[str, tuple[str, bool]] = { 

1182 "This trial": ("trial", False), 

1183 "All": ("all", True), 

1184 "All filtered trials": ("all", False), 

1185 } 

1186 options_map["All trials of one participant"] = ("participant", False) 

1187 options_map["All trials of one text"] = ("text", False) 

1188 

1189 # Default to the filtered subset (respect what the user narrowed to). 

1190 default_index = 2 

1191 scope_label = st.radio( 

1192 "Trials to include", 

1193 options=list(options_map), 

1194 index=default_index, 

1195 key=f"{key_prefix}_scope", 

1196 # The stored value stays "All"; only what the radio shows says more. 

1197 format_func=lambda label: "All, ignoring filters" if label == "All" else label, 

1198 horizontal=True, 

1199 help="Choose a subset. All ignores active filters.", 

1200 label_visibility="collapsed", 

1201 ) 

1202 scope, export_unfiltered = options_map[scope_label] 

1203 active = combos_all if (export_unfiltered and combos_all is not None) else combos 

1204 

1205 scope_participant: str | None = None 

1206 scope_trial: str | None = None 

1207 scope_text: str | None = None 

1208 text_col = ( 

1209 "unique_text_id" 

1210 if "unique_text_id" in active.columns 

1211 else ("text_id" if "text_id" in active.columns else None) 

1212 ) 

1213 

1214 if scope == "trial" and not active.empty: 

1215 if selected_participant is not None and selected_trial is not None: 

1216 scope_participant = str(selected_participant) 

1217 scope_trial = str(selected_trial) 

1218 else: 

1219 participants = sorted( 

1220 active["participant_id"].dropna().astype(str).unique() 

1221 ) 

1222 scope_participant = panel_field( 

1223 st, 

1224 "selectbox", 

1225 "Participant", 

1226 options=participants, 

1227 key=f"{key_prefix}_scope_pid", 

1228 ) 

1229 trials_for_pid = ( 

1230 active.loc[ 

1231 active["participant_id"].astype(str) == str(scope_participant), 

1232 "trial_id", 

1233 ] 

1234 .astype(str) 

1235 .unique() 

1236 ) 

1237 scope_trial = panel_field( 

1238 st, 

1239 "selectbox", 

1240 "Trial", 

1241 options=sorted(trials_for_pid), 

1242 key=f"{key_prefix}_scope_trial", 

1243 ) 

1244 elif scope == "participant" and not active.empty: 

1245 participants = sorted(active["participant_id"].dropna().astype(str).unique()) 

1246 scope_participant = panel_field( 

1247 st, 

1248 "selectbox", 

1249 "Participant", 

1250 options=participants, 

1251 key=f"{key_prefix}_scope_pid", 

1252 ) 

1253 elif scope == "text" and not active.empty: 

1254 if text_col is None: 

1255 st.info("This dataset has no text ids, so it can't be exported by text.") 

1256 else: 

1257 texts = sorted(active[text_col].dropna().astype(str).unique()) 

1258 scope_text = panel_field( 

1259 st, 

1260 "selectbox", 

1261 "Text", 

1262 options=texts, 

1263 key=f"{key_prefix}_scope_text", 

1264 ) 

1265 

1266 # Close the Scope section with a live count of what will be exported. 

1267 n_export = len( 

1268 _scope_frame(active, scope, scope_participant, scope_trial, scope_text) 

1269 ) 

1270 n_total = len(combos_all) if combos_all is not None else len(combos) 

1271 st.caption(f"**{n_export:,}** of **{n_total:,}** trials will be exported.") 

1272 

1273 return scope, scope_participant, scope_trial, scope_text, export_unfiltered 

1274 

1275 

1276def _preview_fields(combos: pd.DataFrame) -> dict: 

1277 """Stand-in field values for the live pattern preview (EXP-1/EXP-2). 

1278 

1279 Uses the first trial in scope, so the preview is a path the user will 

1280 actually get rather than a made-up example. 

1281 """ 

1282 row = combos.iloc[0].to_dict() if combos is not None and not combos.empty else {} 

1283 fields = dict(row) 

1284 fields.setdefault("participant_id", "p01") 

1285 fields.setdefault("trial_id", "t01") 

1286 fields.setdefault("text_id", fields["trial_id"]) 

1287 fields.update( 

1288 n_fixations=123, 

1289 n_words=45, 

1290 reading_time_s=18.4, 

1291 settings="layers: fixations, saccades, text", 

1292 ) 

1293 return fields 

1294 

1295 

1296def _prune_stale_fields(state, state_key: str, names: list[str]) -> None: 

1297 """Drop field names the attached table no longer has from a stored picker 

1298 selection — the house `controls._drop_stale_multi` pattern. 

1299 

1300 Two states must stay apart. An **empty** stored list is the user clearing 

1301 the picker, which leaves the table out of the bundle, so it is kept as is. 

1302 A non-empty list that names **only** stale fields is what a replaced table 

1303 leaves behind; Streamlit would filter it to `[]` and read it as that same 

1304 omission, so it is removed and the picker starts again on every field. 

1305 """ 

1306 stored = state.get(state_key) 

1307 if not isinstance(stored, (list, tuple)) or not stored: 

1308 return 

1309 kept = [name for name in stored if name in names] 

1310 if not kept: 

1311 state.pop(state_key, None) 

1312 elif list(stored) != kept: 

1313 state[state_key] = kept 

1314 

1315 

1316def _render_metadata_field_picker(key_prefix: str): 

1317 """DATA-20 milestone 10 — which participant fields ride along in the bundle. 

1318 

1319 Renders nothing when no table is attached, and returns ``None`` — "no 

1320 restriction" — while every field is still selected, so an export made 

1321 without touching this control is byte-identical to one made before the 

1322 control existed. 

1323 

1324 It lives with the export options rather than beside the table on the 🗂️ Data 

1325 page because it is a decision about *this bundle*: the same attached table 

1326 can reasonably ship its full detail to a collaborator and only a group label 

1327 to a public repository. 

1328 """ 

1329 import streamlit as st 

1330 

1331 from scanpath_studio import metadata as md 

1332 

1333 attached = md.active() 

1334 if attached is None or not attached.fields: 

1335 return None 

1336 names = [field.name for field in attached.fields] 

1337 labels = {field.name: field.label for field in attached.fields} 

1338 state_key = f"{key_prefix}_meta_fields" 

1339 _prune_stale_fields(st.session_state, state_key, names) 

1340 chosen = panel_field( 

1341 st, 

1342 "multiselect", 

1343 "Participant fields to include", 

1344 options=names, 

1345 default=names, 

1346 format_func=lambda name: labels.get(name, name), 

1347 key=state_key, 

1348 persist_state="session", 

1349 help="The participant id is always kept.", 

1350 ) 

1351 ordered = tuple(name for name in names if name in set(chosen)) 

1352 return None if len(ordered) == len(names) else ordered 

1353 

1354 

1355def _render_trial_metadata_field_picker(key_prefix: str): 

1356 """DATA-29 — which trial fields ride along, the twin of the picker above. 

1357 

1358 Same contract in every respect: nothing rendered and ``None`` returned when 

1359 no trial table is attached or while every field is still chosen, so an 

1360 export made without touching it is unchanged. 

1361 """ 

1362 import streamlit as st 

1363 

1364 from scanpath_studio import metadata as md 

1365 

1366 attached = md.active_trials() 

1367 if attached is None or not attached.fields: 

1368 return None 

1369 names = [field.name for field in attached.fields] 

1370 labels = {field.name: field.label for field in attached.fields} 

1371 state_key = f"{key_prefix}_trial_meta_fields" 

1372 _prune_stale_fields(st.session_state, state_key, names) 

1373 chosen = panel_field( 

1374 st, 

1375 "multiselect", 

1376 "Trial fields to include", 

1377 options=names, 

1378 default=names, 

1379 format_func=lambda name: labels.get(name, name), 

1380 key=state_key, 

1381 persist_state="session", 

1382 help="The trial id is always kept.", 

1383 ) 

1384 ordered = tuple(name for name in names if name in set(chosen)) 

1385 return None if len(ordered) == len(names) else ordered 

1386 

1387 

1388def _render_text_metadata_field_picker(key_prefix: str): 

1389 """Which text fields ride along — third grain, twin of the two above. 

1390 

1391 Same contract: nothing rendered and ``None`` returned when no text table 

1392 is attached or while every field is still chosen, so an export made 

1393 without touching it is unchanged. 

1394 """ 

1395 import streamlit as st 

1396 

1397 from scanpath_studio import metadata as md 

1398 

1399 attached = md.active_texts() 

1400 if attached is None or not attached.fields: 

1401 return None 

1402 names = [field.name for field in attached.fields] 

1403 labels = {field.name: field.label for field in attached.fields} 

1404 state_key = f"{key_prefix}_text_meta_fields" 

1405 _prune_stale_fields(st.session_state, state_key, names) 

1406 chosen = panel_field( 

1407 st, 

1408 "multiselect", 

1409 "Text fields to include", 

1410 options=names, 

1411 default=names, 

1412 format_func=lambda name: labels.get(name, name), 

1413 key=state_key, 

1414 persist_state="session", 

1415 help="The text id is always kept.", 

1416 ) 

1417 ordered = tuple(name for name in names if name in set(chosen)) 

1418 return None if len(ordered) == len(names) else ordered 

1419 

1420 

1421def _render_naming_options(st, combos: pd.DataFrame, key_prefix: str): 

1422 """The compact **File naming** block: EXP-1's path pattern. 

1423 

1424 Every pattern is validated and previewed against the first trial in scope as 

1425 it's typed — finding a typo after a 200-trial render is the worst possible 

1426 place to find it. Returns ``path_pattern``; an invalid pattern falls back to 

1427 its default so a bad keystroke can't produce a broken zip. 

1428 

1429 The title/caption pair used to live here too (EXP-2) but moved to the 

1430 Scanpath rail's **📐 Figure & canvas** group (EXP-5), so it's visible on the 

1431 live figure and not just at export time; `render_export_options` reads it 

1432 back from there instead of keeping a second, possibly-diverging copy. 

1433 """ 

1434 fields = _preview_fields(combos) 

1435 available = ", ".join( 

1436 f"`{{{k}}}`" for k in sorted(set(fields) | set(_PATTERN_EXTRA_FIELDS)) 

1437 ) 

1438 

1439 def _pattern_input(label: str, default: str, key: str, help_text: str) -> str: 

1440 value = panel_field( 

1441 st, "text_input", label, value=default, key=key, help=help_text 

1442 ) 

1443 error = pattern_error(value, fields) or path_structure_error(value) 

1444 if error: 

1445 st.error(error) 

1446 return default 

1447 return value 

1448 

1449 heading, fields_slot = st.columns([5, 1], vertical_alignment="bottom") 

1450 heading.markdown("### File naming") 

1451 with fields_slot.popover( 

1452 "Fields", help="Placeholders available in the path pattern." 

1453 ): 

1454 st.markdown(available) 

1455 path_pattern = _pattern_input( 

1456 "File path pattern", 

1457 DEFAULT_PATH_PATTERN, 

1458 f"{key_prefix}_path_pattern", 

1459 "Path inside the ZIP. Use `/` for folders and `{…}` placeholders.", 

1460 ) 

1461 st.caption( 

1462 "Example: `" 

1463 + resolve_export_path( 

1464 path_pattern, fields, artifact="figure", ext="png", used=set() 

1465 ) 

1466 + "`" 

1467 ) 

1468 return path_pattern 

1469 

1470 

1471def render_export_options( 

1472 st_module, 

1473 combos: pd.DataFrame, 

1474 key_prefix: str = "export", 

1475 combos_all: pd.DataFrame | None = None, 

1476 title_pattern: str = "", 

1477 caption_pattern: str = "", 

1478 selected_participant: str | None = None, 

1479 selected_trial: str | None = None, 

1480 screen_options: list[str] | None = None, 

1481) -> ExportOptions: 

1482 """Render the bulk-export options UI and return a populated ExportOptions. 

1483 

1484 ``screen_options`` are the dataset's screen ids (:func:`screen_choices`); 

1485 when there are any, a *Screens* picker narrows the bundle to some of them. 

1486 

1487 ``combos`` is the currently filtered trial pool; ``combos_all`` (when given) 

1488 is the whole loaded dataset. Picking the "All" scope switches the scope 

1489 picker — and the export itself — to ``combos_all`` so the trial filters 

1490 are ignored. ``title_pattern``/``caption_pattern`` come from the Scanpath 

1491 rail's **📐 Figure & canvas** → *Title* / *Caption* (EXP-5) — 

1492 this panel no longer has its own copy of that setting. 

1493 """ 

1494 st = st_module 

1495 # No expander — the options are always displayed. 

1496 with st.container(): 

1497 st.markdown("### Trials to include") 

1498 # The whole-dataset choice lives inside the scope radio. 

1499 ( 

1500 scope, 

1501 scope_pid, 

1502 scope_trial, 

1503 scope_text, 

1504 export_unfiltered, 

1505 ) = _render_scope_picker( 

1506 st, 

1507 combos, 

1508 key_prefix, 

1509 combos_all=combos_all, 

1510 selected_participant=selected_participant, 

1511 selected_trial=selected_trial, 

1512 ) 

1513 screens = _render_screen_picker(st, screen_options or [], key_prefix) 

1514 

1515 # Figures are the headline artifact, so they lead with a single 

1516 # multi-select of formats (pills) rather than a column of checkboxes. 

1517 st.markdown("### Figure formats") 

1518 fig_formats = ( 

1519 panel_field( 

1520 st, 

1521 "pills", 

1522 "Formats", 

1523 options=["PDF", "SVG", "PNG", "HTML"], 

1524 selection_mode="multi", 

1525 default=["PDF"], 

1526 key=f"{key_prefix}_figfmts", 

1527 help="PDF/SVG are vector, PNG is raster, and HTML is interactive. " 

1528 "The bundle draws PDF, SVG and PNG with Chrome, Chromium or " 

1529 "Edge; HTML needs no browser.", 

1530 ) 

1531 or [] 

1532 ) 

1533 include_pdf = "PDF" in fig_formats 

1534 include_svg = "SVG" in fig_formats 

1535 include_png = "PNG" in fig_formats 

1536 include_html = "HTML" in fig_formats 

1537 # Asked only while HTML is picked, as for the current figure above. 

1538 html_self_contained = include_html and bool( 

1539 panel_field( 

1540 st, 

1541 "checkbox", 

1542 "Self-contained HTML (opens offline, larger file)", 

1543 display="Self-contained HTML", 

1544 value=False, 

1545 key=f"{key_prefix}_html_self_contained", 

1546 persist_state="session", 

1547 help="Tick to make the bundle's HTML figures open without an " 

1548 "internet connection: each carries the Plotly library, about " 

1549 "4.8 MB more. Unticked, they load it from cdn.plot.ly when opened.", 

1550 ) 

1551 ) 

1552 # EXP-24: the bundle's static figures need a browser on the server, 

1553 # which the current figure's PNG/SVG do not — said only when it is 

1554 # missing, and only while one of those formats is picked. 

1555 if note := missing_browser_note(include_pdf or include_svg or include_png): 

1556 st.warning(note, icon=ICONS["warning"]) 

1557 # Only surface the scale stepper when PNG is on, and keep it narrow. 

1558 # `width` does here what the old `st.columns([1, 3])` did: a 1–4 stepper 

1559 # stretched across the whole field column reads as a text box. UX-69 

1560 # dropped the columns because they nested inside the row's own. 

1561 if include_png: 

1562 png_scale = panel_field( 

1563 st, 

1564 "number_input", 

1565 "PNG scale", 

1566 min_value=1, 

1567 max_value=4, 

1568 value=2, 

1569 width=140, 

1570 key=f"{key_prefix}_scale", 

1571 help="Higher values increase quality and file size.", 

1572 ) 

1573 else: 

1574 png_scale = int(st.session_state.get(f"{key_prefix}_scale", 2)) 

1575 

1576 st.markdown("### Also include") 

1577 # VIZ-5: per-layer figure breakdown for publication editing. 

1578 separable_layers = panel_field( 

1579 st, 

1580 "toggle", 

1581 "Separable layers", 

1582 value=False, 

1583 key=f"{key_prefix}_layers", 

1584 help="Export each visual layer as a separate file.", 

1585 ) 

1586 # Layers are static vectors/rasters — HTML can't be split. When no 

1587 # vector/raster format is picked, they fall back to SVG (which needs 

1588 # Kaleido/Chrome, unlike the browser-free HTML the user chose), so warn. 

1589 if separable_layers and not (include_png or include_svg or include_pdf): 

1590 st.caption( 

1591 f"{ICONS['warning']} Separable layers export as **SVG** (a static vector needing " 

1592 "Chrome/Kaleido) — HTML figures can't be split. Pick SVG/PDF/PNG " 

1593 "above to choose the layer format." 

1594 ) 

1595 include_plot_config = panel_field( 

1596 st, 

1597 "toggle", 

1598 "Settings file (JSON)", 

1599 value=True, 

1600 key=f"{key_prefix}_cfg", 

1601 help="Include the figure's settings file.", 

1602 ) 

1603 include_annotations = panel_field( 

1604 st, 

1605 "toggle", 

1606 "Annotations (JSON)", 

1607 value=False, 

1608 key=f"{key_prefix}_annotations", 

1609 help="Include the exported trials' favorites, tags and notes as one " 

1610 f"annotations.json — the file {ICONS['view_data']} Data Management → Annotations imports.", 

1611 ) 

1612 tabular = ( 

1613 panel_field( 

1614 st, 

1615 "pills", 

1616 "Tabular data", 

1617 # The measure family is computed by the app, and held back 

1618 # with the other computed measures. 

1619 options=[ 

1620 "Fixations", 

1621 "Raw gaze", 

1622 "Word measures", 

1623 *(["Full measure family"] if computed_measures_enabled() else []), 

1624 ], 

1625 selection_mode="multi", 

1626 default=[], 

1627 key=f"{key_prefix}_tabular", 

1628 help="Choose the data tables to include. Word measures are the " 

1629 "reading measures the dataset brought; none are computed.", 

1630 ) 

1631 or [] 

1632 ) 

1633 include_fixations = "Fixations" in tabular 

1634 include_raw_gaze = "Raw gaze" in tabular 

1635 include_measures = "Word measures" in tabular 

1636 include_analysis_family = "Full measure family" in tabular 

1637 any_table = bool(tabular) 

1638 if any_table: 

1639 table_format = ( 

1640 panel_field( 

1641 st, 

1642 "segmented_control", 

1643 "Table format", 

1644 options=["csv", "parquet", "both"], 

1645 format_func=lambda value: { 

1646 "csv": "CSV", 

1647 "parquet": "Parquet", 

1648 "both": "Both", 

1649 }[value], 

1650 default="csv", 

1651 key=f"{key_prefix}_fmt", 

1652 ) 

1653 or "csv" 

1654 ) 

1655 combine_trials = panel_field( 

1656 st, 

1657 "toggle", 

1658 "Combine all trials into one file per table", 

1659 value=False, 

1660 key=f"{key_prefix}_combine", 

1661 help="Write each table once, with every exported trial stacked " 

1662 "in it, under aggregate/ — instead of one file per trial.", 

1663 ) 

1664 else: 

1665 table_format = str(st.session_state.get(f"{key_prefix}_fmt", "csv")) 

1666 combine_trials = bool(st.session_state.get(f"{key_prefix}_combine", False)) 

1667 

1668 metadata_fields = _render_metadata_field_picker(key_prefix) 

1669 trial_metadata_fields = _render_trial_metadata_field_picker(key_prefix) 

1670 text_metadata_fields = _render_text_metadata_field_picker(key_prefix) 

1671 path_pattern = _render_naming_options(st, combos, key_prefix) 

1672 if title_pattern or caption_pattern: 

1673 st.caption( 

1674 "Title and caption on the figure — set on the Scanpath rail's " 

1675 f"**{ICONS['figure']} Figure & canvas** → *Title* / *Caption*, and " 

1676 "applied here too." 

1677 ) 

1678 

1679 return ExportOptions( 

1680 include_png=include_png, 

1681 include_svg=include_svg, 

1682 include_pdf=include_pdf, 

1683 include_html=include_html, 

1684 html_self_contained=html_self_contained, 

1685 include_plot_config=include_plot_config, 

1686 include_annotations=include_annotations, 

1687 include_fixations=include_fixations, 

1688 include_raw_gaze=include_raw_gaze, 

1689 include_measures=include_measures, 

1690 include_analysis_family=include_analysis_family, 

1691 combine_trials=combine_trials, 

1692 separable_layers=separable_layers, 

1693 table_format=table_format, 

1694 png_scale=int(png_scale), 

1695 path_pattern=path_pattern, 

1696 title_pattern=title_pattern, 

1697 caption_pattern=caption_pattern, 

1698 # VIZ-36: this panel only runs inside the app, so the picker's label is 

1699 # available; `pattern_fields` itself stays session-free. 

1700 dataset_name=_session_dataset_name(), 

1701 metadata_fields=metadata_fields, 

1702 trial_metadata_fields=trial_metadata_fields, 

1703 text_metadata_fields=text_metadata_fields, 

1704 export_unfiltered=export_unfiltered, 

1705 scope=scope, 

1706 scope_participant=scope_pid, 

1707 scope_trial=scope_trial, 

1708 scope_text=scope_text, 

1709 screens=screens, 

1710 ) 

1711 

1712 

1713def _render_screen_picker( 

1714 st, screen_options: list[str], key_prefix: str 

1715) -> tuple[str, ...] | None: 

1716 """The *Screens* row: which screens of each multipart trial go in. Nothing 

1717 picked is every screen (``None``); drawn only for a dataset with screens.""" 

1718 if not screen_options: 

1719 return None 

1720 key = f"{key_prefix}_screens" 

1721 # A pick from another dataset names screens this one does not have. 

1722 held = st.session_state.get(key) 

1723 if held is not None: 

1724 kept = [screen for screen in held if screen in screen_options] 

1725 if kept != list(held): 

1726 st.session_state[key] = kept 

1727 picked = ( 

1728 panel_field( 

1729 st, 

1730 "multiselect", 

1731 "Screens", 

1732 options=screen_options, 

1733 key=key, 

1734 placeholder="All screens", 

1735 help="Export only these screens of each trial. Leave empty for " 

1736 "every screen; a trial that shows none of them is left out.", 

1737 ) 

1738 or [] 

1739 ) 

1740 return tuple(picked) if picked else None 

1741 

1742 

1743def _session_dataset_name() -> str: 

1744 """The dataset picker's label, for the options UI only (VIZ-36). 

1745 

1746 Guarded because `export` is imported headless by `api` and `cli`, where 

1747 there is no session; the value only ever reaches `ExportOptions`, never 

1748 `pattern_fields`, which takes it as an argument by design. 

1749 """ 

1750 import streamlit as st 

1751 

1752 try: 

1753 return str(st.session_state.get("data_source_choice") or "") 

1754 except Exception: 

1755 return "" 

1756 

1757 

1758def _scope_frame( 

1759 combos: pd.DataFrame, 

1760 scope: str, 

1761 scope_participant: str | None, 

1762 scope_trial: str | None, 

1763 scope_text: str | None, 

1764) -> pd.DataFrame: 

1765 """Filter combos to the chosen scope (pure helper, no ExportOptions needed).""" 

1766 if scope == "trial" and scope_participant and scope_trial: 

1767 return combos[ 

1768 (combos["participant_id"].astype(str) == str(scope_participant)) 

1769 & (combos["trial_id"].astype(str) == str(scope_trial)) 

1770 ] 

1771 if scope == "participant" and scope_participant: 

1772 return combos[combos["participant_id"].astype(str) == str(scope_participant)] 

1773 if scope == "text" and scope_text: 

1774 text_col = ( 

1775 "unique_text_id" 

1776 if "unique_text_id" in combos.columns 

1777 else ("text_id" if "text_id" in combos.columns else None) 

1778 ) 

1779 if text_col is None: 

1780 return combos 

1781 return combos[combos[text_col].astype(str) == str(scope_text)] 

1782 return combos 

1783 

1784 

1785def _apply_scope(combos: pd.DataFrame, options: ExportOptions) -> pd.DataFrame: 

1786 """Filter combos according to options.scope.""" 

1787 return _scope_frame( 

1788 combos, 

1789 options.scope, 

1790 options.scope_participant, 

1791 options.scope_trial, 

1792 options.scope_text, 

1793 ) 

1794 

1795 

1796@dataclass(frozen=True) 

1797class ComparisonSide: 

1798 """One half of an exported comparison pair (CMP-8 §6). 

1799 

1800 ``dataset`` is the corpus label — ``None`` for the active dataset, which is 

1801 what every same-dataset comparison passes. ``participant`` / ``trial`` are 

1802 the **real** ids, as the corpus spells them: the ``dataset · pid`` namespace 

1803 exists only inside the figure's throwaway frames, and an exported table that 

1804 used it would not match its own corpus. 

1805 """ 

1806 

1807 participant: str 

1808 trial: str 

1809 words: pd.DataFrame 

1810 fixations: pd.DataFrame 

1811 dataset: str | None = None 

1812 setup: dict | None = None 

1813 

1814 @property 

1815 def slug(self) -> str: 

1816 return f"{_safe_id(self.participant)}__{_safe_id(self.trial)}" 

1817 

1818 def stamped(self, side: str | None = None) -> tuple[pd.DataFrame, pd.DataFrame]: 

1819 """Both frames with a ``dataset`` column, so the pair's tables are readable. 

1820 

1821 Two corpora can hold the same ``(participant_id, trial_id)``; without 

1822 this column the rows in ``fixations.csv`` would be indistinguishable. 

1823 ``side`` (CMP-22) also stamps a ``scanpath`` column, ``"A"`` or ``"B"``: 

1824 B can now be A's own trial, whose rows match A's on every other column. 

1825 """ 

1826 label = self.dataset or "(this dataset)" 

1827 out = [] 

1828 for frame in (self.words, self.fixations): 

1829 if frame is None or frame.empty: 

1830 out.append(frame) 

1831 continue 

1832 stamped = frame.copy() 

1833 stamped["dataset"] = label 

1834 if side is not None: 

1835 stamped["scanpath"] = side 

1836 out.append(stamped) 

1837 return out[0], out[1] 

1838 

1839 def manifest(self) -> dict: 

1840 return { 

1841 "source": self.dataset, 

1842 "participant": str(self.participant), 

1843 "trial": str(self.trial), 

1844 "setup": self.setup, 

1845 } 

1846 

1847 

1848def pair_export( 

1849 fig, 

1850 side_a: ComparisonSide, 

1851 side_b: ComparisonSide, 

1852 *, 

1853 canvas_width: int, 

1854 canvas_height: int, 

1855 x_field: str, 

1856 y_field: str, 

1857 settings: dict, 

1858 options: ExportOptions, 

1859 status_callback: StatusCallback | None = None, 

1860 column_names: Mapping[str, ColumnNames] | None = None, 

1861) -> bytes: 

1862 """Zip one comparison **pair** — figure, manifest, and both scanpaths' tables. 

1863 

1864 ``column_names`` (DATA-66) is the active dataset's map per table. It names 

1865 the tables' columns only when both scanpaths come from that dataset: a pair 

1866 across two datasets shares no file names, so its tables keep the internal 

1867 ones, which both datasets understand. 

1868 

1869 CMP-8 §6. An exported cross-dataset figure is unreproducible on its own: 

1870 nothing in the image records where B came from. The bundle writes the bulk 

1871 exporter's per-trial shape for the pair instead of for one trial, and the 

1872 ``datasets`` block in ``plot_config.json`` is what makes it reproducible — 

1873 it names both sources, both trials, and both recording setups:: 

1874 

1875 <A>__vs__<B>/ 

1876 ├─ figure.<fmt> 

1877 ├─ plot_config.json 

1878 ├─ fixations.csv 

1879 └─ measures.csv 

1880 

1881 ``options`` is the ordinary `ExportOptions` — the pair is one more "trial 

1882 folder" as far as the writer is concerned, so the formats, table format and 

1883 figure toggles all mean what they already mean. Deliberately unchanged from 

1884 bulk export: the tables stay **uncorrected** by drift correction (EXP-4 / 

1885 VIZ-24), which the manifest records. 

1886 """ 

1887 folder = f"{side_a.slug}__vs__{side_b.slug}" 

1888 buffer = io.BytesIO() 

1889 started = perf_counter() 

1890 emit_status( 

1891 status_callback, 

1892 ExportStage.PREPARING, 

1893 "Preparing the comparison pair…", 

1894 started_at=started, 

1895 ) 

1896 with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as zf: 

1897 for fmt in options.figure_formats(): 

1898 if fig is None: 

1899 break 

1900 width = int(getattr(fig.layout, "width", None) or canvas_width) 

1901 height = int(getattr(fig.layout, "height", None) or canvas_height) 

1902 if fmt == "html": 

1903 data = fig.to_html( 

1904 include_plotlyjs=html_plotlyjs(options.html_self_contained), 

1905 full_html=True, 

1906 config={**PLOTLY_CONFIG}, 

1907 ).encode("utf-8") 

1908 else: 

1909 data = render_static_figure_bytes( 

1910 fig, 

1911 fmt=fmt, 

1912 width=width, 

1913 height=height, 

1914 scale=3 if fmt == "png" else 1, 

1915 status_callback=status_callback, 

1916 ) 

1917 zf.writestr(f"{folder}/figure.{fmt}", data) 

1918 

1919 words_a, fix_a = side_a.stamped("A") 

1920 words_b, fix_b = side_b.stamped("B") 

1921 maps = ( 

1922 { 

1923 table: names 

1924 for table, names in (column_names or {}).items() 

1925 if names is not None and names.entries 

1926 } 

1927 if side_b.dataset in (None, side_a.dataset) 

1928 else {} 

1929 ) 

1930 # AN-32 / EXP-23: the measures each side's dataset brought. 

1931 measures = [ 

1932 words.assign(scanpath=side) 

1933 for side, words in (("A", words_a), ("B", words_b)) 

1934 if words is not None and not words.empty and brought_reading_measures(words) 

1935 ] 

1936 tables = { 

1937 "fixations": pd.concat([fix_a, fix_b], ignore_index=True) 

1938 if options.include_fixations 

1939 else None, 

1940 "words": pd.concat(measures, ignore_index=True) 

1941 if options.include_measures and measures 

1942 else None, 

1943 } 

1944 files = {"fixations": "fixations", "words": "measures"} 

1945 written = { 

1946 table: written_columns(shareable_frame(frame), maps[table]) 

1947 for table, frame in tables.items() 

1948 if frame is not None and table in maps 

1949 } 

1950 for fmt in options.table_formats(): 

1951 for table, frame in tables.items(): 

1952 if frame is not None: 

1953 _write_table( 

1954 zf, 

1955 f"{folder}/{files[table]}.{fmt}", 

1956 frame, 

1957 fmt, 

1958 maps.get(table), 

1959 ) 

1960 if any(written.values()): 

1961 zf.writestr( 

1962 f"{folder}/columns.json", 

1963 json.dumps(columns_manifest(written), indent=2), 

1964 ) 

1965 

1966 config = _plot_config_dict( 

1967 side_a.participant, 

1968 side_a.trial, 

1969 canvas_width, 

1970 canvas_height, 

1971 x_field, 

1972 y_field, 

1973 settings, 

1974 ) 

1975 # The block that makes the pair reproducible — without it the figure 

1976 # names two readers and no way to find either of them again. 

1977 config["datasets"] = {"a": side_a.manifest(), "b": side_b.manifest()} 

1978 zf.writestr( 

1979 f"{folder}/plot_config.json", json.dumps(config, indent=2).encode("utf-8") 

1980 ) 

1981 emit_status( 

1982 status_callback, 

1983 ExportStage.READY, 

1984 "Comparison pair ready.", 

1985 started_at=started, 

1986 ) 

1987 return buffer.getvalue() 

1988 

1989 

1990def _selected_metadata_columns(frame, fields: tuple[str, ...] | None): 

1991 """``frame`` narrowed to ``fields`` (+ the reader id), or ``None`` to drop it. 

1992 

1993 ``fields is None`` keeps every column — the default, so nothing changes for 

1994 a caller that never heard of the opt-out. An **empty** tuple means the user 

1995 cleared the picker: the table is left out of the bundle entirely, rather 

1996 than shipped as a bare list of reader ids. 

1997 """ 

1998 if frame is None or fields is None: 

1999 return frame 

2000 if not fields: 

2001 return None 

2002 keep = ["participant_id"] if "participant_id" in frame.columns else [] 

2003 keep += [name for name in fields if name in frame.columns and name not in keep] 

2004 return frame[keep] if keep else None 

2005 

2006 

2007def _selected_trial_metadata_columns(frame, fields: tuple[str, ...] | None): 

2008 """``frame`` narrowed to ``fields`` (+ the trial key), or ``None`` to drop it. 

2009 

2010 The trial twin of :func:`_selected_metadata_columns`. The key it always 

2011 keeps is whichever the table was attached with — ``trial_id`` alone, or 

2012 ``participant_id`` beside it — because a trial table shipped without its 

2013 key cannot be joined back to anything. 

2014 """ 

2015 if frame is None or fields is None: 

2016 return frame 

2017 if not fields: 

2018 return None 

2019 keep = [name for name in ("participant_id", "trial_id") if name in frame.columns] 

2020 keep += [name for name in fields if name in frame.columns and name not in keep] 

2021 return frame[keep] if keep else None 

2022 

2023 

2024def _selected_text_metadata_columns(frame, fields: tuple[str, ...] | None): 

2025 """``frame`` narrowed to ``fields`` (+ the text key), or ``None`` to drop it. 

2026 

2027 The text twin of :func:`_selected_metadata_columns`/ 

2028 :func:`_selected_trial_metadata_columns` — a text table shipped without 

2029 its key cannot be joined back to anything. 

2030 """ 

2031 if frame is None or fields is None: 

2032 return frame 

2033 if not fields: 

2034 return None 

2035 keep = ["text_id"] if "text_id" in frame.columns else [] 

2036 keep += [name for name in fields if name in frame.columns and name not in keep] 

2037 return frame[keep] if keep else None 

2038 

2039 

2040def _rows_in_scope( 

2041 frame, 

2042 *, 

2043 pairs: set[tuple[str, str]], 

2044 texts: set[str], 

2045 grain: str, 

2046): 

2047 """``frame``'s rows about what the bundle exports, or ``None`` when none are. 

2048 

2049 A bundle's metadata describes only its own readers, trials and texts: 

2050 ``grain`` ``"participant"`` keeps the readers in ``pairs``, ``"trial"`` the 

2051 readings — by the (reader, trial) pair when the table has a reader column, 

2052 else by trial id — and ``"text"`` the texts in ``texts``. Ids compare as 

2053 text, as the metadata module matches them. A table with no key column, or 

2054 no matching row, is left out rather than shipped whole. 

2055 """ 

2056 if frame is None or frame.empty: 

2057 return None 

2058 if grain == "participant": 

2059 if "participant_id" not in frame.columns: 

2060 return None 

2061 mask = frame["participant_id"].astype(str).isin({p for p, _ in pairs}) 

2062 elif grain == "trial": 

2063 if "trial_id" not in frame.columns: 

2064 return None 

2065 if "participant_id" in frame.columns: 

2066 keys = pd.Series( 

2067 list( 

2068 zip( 

2069 frame["participant_id"].astype(str), 

2070 frame["trial_id"].astype(str), 

2071 strict=True, 

2072 ) 

2073 ), 

2074 index=frame.index, 

2075 dtype=object, 

2076 ) 

2077 mask = keys.isin(pairs) 

2078 else: 

2079 mask = frame["trial_id"].astype(str).isin({t for _, t in pairs}) 

2080 else: 

2081 if "text_id" not in frame.columns: 

2082 return None 

2083 mask = frame["text_id"].astype(str).isin(texts) 

2084 kept = frame[mask] 

2085 return None if kept.empty else kept 

2086 

2087 

2088def _unit_text_ids(combo_row: dict, *frames: pd.DataFrame) -> set[str]: 

2089 """Every text id one exported reading carries — its combo row's, and the 

2090 ``unique_text_id`` / ``text_id`` values in its own rows — as text.""" 

2091 found: set[str] = set() 

2092 value = combo_row.get("text_id") 

2093 if value is not None and not pd.isna(value): 

2094 found.add(str(value)) 

2095 for frame in frames: 

2096 if frame is None or frame.empty: 

2097 continue 

2098 for column in ("unique_text_id", "text_id"): 

2099 if column in frame.columns: 

2100 found.update(frame[column].dropna().astype(str).unique()) 

2101 return found 

2102 

2103 

2104def _session_text_metadata(): 

2105 """The attached text table, when running inside the app. 

2106 

2107 Handed over on a private session key for the same reason its two 

2108 siblings are: the frame must not reach the bulk-export cache signature. 

2109 """ 

2110 try: 

2111 import streamlit as st 

2112 

2113 return st.session_state.get("_export_text_metadata") 

2114 except Exception: # no script run context (API, CLI) 

2115 return None 

2116 

2117 

2118def _session_trial_metadata(): 

2119 """The attached trial table, when running inside the app (DATA-29). 

2120 

2121 Handed over on a private session key for the same reason its participant 

2122 twin is: the frame must not reach the bulk-export cache signature. 

2123 """ 

2124 try: 

2125 import streamlit as st 

2126 

2127 return st.session_state.get("_export_trial_metadata") 

2128 except Exception: # no script run context (API, CLI) 

2129 return None 

2130 

2131 

2132def _session_participant_metadata(): 

2133 """The attached participant table, when running inside the app (DATA-20). 

2134 

2135 Handed over through session state rather than through ``settings`` so the 

2136 frame never reaches the bulk-export cache signature, which stringifies the 

2137 settings dict and would truncate a long table into a colliding key. Returns 

2138 ``None`` headlessly, where callers pass the frame in ``settings`` instead. 

2139 """ 

2140 try: 

2141 import streamlit as st 

2142 

2143 return st.session_state.get("_export_participant_metadata") 

2144 except Exception: # no script run context (API, CLI) 

2145 return None 

2146 

2147 

2148#: `index.csv`'s columns, in order. 

2149INVENTORY_COLUMNS = ( 

2150 "path", 

2151 "artifact", 

2152 "format", 

2153 "participant_id", 

2154 "trial_id", 

2155 "screen_id", 

2156 "status", 

2157 "note", 

2158) 

2159 

2160 

2161def _write_inventory(zf: zipfile.ZipFile, inventory: list[dict]) -> None: 

2162 """Write ``index.csv``: one row per file in the bundle, plus each requested 

2163 file that failed and each reading skipped. ``status`` is ``written``, 

2164 ``failed`` or ``skipped``; a file type nobody asked for has no row.""" 

2165 frame = pd.DataFrame(inventory, columns=list(INVENTORY_COLUMNS)) 

2166 zf.writestr("index.csv", frame.to_csv(index=False)) 

2167 

2168 

2169def screen_choices(*frames: pd.DataFrame | None) -> list[str]: 

2170 """The screen ids ``frames`` carry, for the export's screen picker: in the 

2171 order they are shown (their lowest ``screen_index``), then by id. Empty when 

2172 no frame has screens.""" 

2173 parts = [ 

2174 # De-duplicated first: a raw-gaze table can hold millions of samples. 

2175 frame[ 

2176 [c for c in (SCREEN_ID, SCREEN_INDEX) if c in frame.columns] 

2177 ].drop_duplicates() 

2178 for frame in frames 

2179 if frame is not None and not frame.empty and SCREEN_ID in frame.columns 

2180 ] 

2181 if not parts: 

2182 return [] 

2183 pairs = pd.concat(parts, ignore_index=True).dropna(subset=[SCREEN_ID]) 

2184 pairs[SCREEN_ID] = pairs[SCREEN_ID].astype(str) 

2185 if SCREEN_INDEX not in pairs.columns: 

2186 pairs[SCREEN_INDEX] = float("nan") 

2187 pairs[SCREEN_INDEX] = pd.to_numeric(pairs[SCREEN_INDEX], errors="coerce") 

2188 first = pairs.groupby(SCREEN_ID)[SCREEN_INDEX].min().reset_index() 

2189 first = first.sort_values([SCREEN_INDEX, SCREEN_ID], na_position="last") 

2190 return first[SCREEN_ID].tolist() 

2191 

2192 

2193def export_units( 

2194 combos: pd.DataFrame, 

2195 words: pd.DataFrame, 

2196 fixations: pd.DataFrame, 

2197 raw_gaze: pd.DataFrame | None = None, 

2198 screens: tuple[str, ...] | None = None, 

2199) -> pd.DataFrame: 

2200 """One row per **screen export unit**: each trial in ``combos``, or each of 

2201 its screens when it is a multipart reading. ``combos`` is already scoped. 

2202 

2203 ``screens`` (``ExportOptions.screens``) keeps only the units on those 

2204 screen ids, so a trial without screens has none of them and is left out. 

2205 """ 

2206 rows: list[dict] = [] 

2207 for combo in combos.to_dict("records"): 

2208 participant, trial = combo["participant_id"], combo["trial_id"] 

2209 parent_words = extract_trial(words, participant, trial) 

2210 parent_fixations = extract_trial(fixations, participant, trial) 

2211 catalog = part_catalog(parent_words, parent_fixations) 

2212 if ( 

2213 catalog.empty 

2214 and parent_words.empty 

2215 and parent_fixations.empty 

2216 and raw_gaze is not None 

2217 and SCREEN_ID in raw_gaze.columns 

2218 ): 

2219 # VIZ-45, as `api._select_part` decides it: a trial recorded as raw 

2220 # gaze alone takes its screens from its samples, so each screen's 

2221 # coordinate space is exported on its own instead of pooled. 

2222 catalog = part_catalog(extract_trial(raw_gaze, participant, trial)) 

2223 if catalog.empty: 

2224 rows.append(combo) 

2225 else: 

2226 for screen in catalog.to_dict("records"): 

2227 rows.append({**combo, **screen}) 

2228 units = pd.DataFrame(rows) 

2229 if screens is None: 

2230 return units 

2231 if units.empty or SCREEN_ID not in units.columns: 

2232 return units.iloc[0:0] 

2233 chosen = {str(screen) for screen in screens} 

2234 keep = units[SCREEN_ID].notna() & units[SCREEN_ID].astype(str).isin(chosen) 

2235 return units[keep].reset_index(drop=True) 

2236 

2237 

2238@dataclass(frozen=True) 

2239class ExportPlan: 

2240 """What a bundle will hold, worked out before it is built. 

2241 

2242 ``layer_files`` is a ceiling: a screen writes one file per layer it 

2243 actually draws, which is only known once its figure is built. 

2244 """ 

2245 

2246 trials: int 

2247 units: int 

2248 figure_files: int 

2249 layer_files: int 

2250 

2251 

2252def apply_export_scope(combos: pd.DataFrame, options: ExportOptions) -> pd.DataFrame: 

2253 """``combos`` narrowed to ``options``' scope, as :func:`bulk_export` does.""" 

2254 return _apply_scope(combos, options) 

2255 

2256 

2257def plan_export( 

2258 combos: pd.DataFrame, 

2259 words: pd.DataFrame, 

2260 fixations: pd.DataFrame, 

2261 options: ExportOptions, 

2262 raw_gaze: pd.DataFrame | None = None, 

2263) -> ExportPlan: 

2264 """The trial, screen and figure-file counts :func:`bulk_export` will 

2265 produce for these inputs and ``options``.""" 

2266 scoped = _apply_scope(combos, options) 

2267 trials, units = count_export(scoped, words, fixations, raw_gaze, options.screens) 

2268 return plan_from_counts(trials, units, options) 

2269 

2270 

2271def count_export_units( 

2272 combos: pd.DataFrame, 

2273 words: pd.DataFrame, 

2274 fixations: pd.DataFrame, 

2275 raw_gaze: pd.DataFrame | None = None, 

2276) -> int: 

2277 """``len(export_units(...))`` — without walking every trial when no frame 

2278 carries screens, the common case, where each trial is one unit.""" 

2279 if not any( 

2280 frame is not None and SCREEN_ID in frame.columns 

2281 for frame in (words, fixations, raw_gaze) 

2282 ): 

2283 return len(combos) 

2284 return len(export_units(combos, words, fixations, raw_gaze)) 

2285 

2286 

2287def count_export( 

2288 combos: pd.DataFrame, 

2289 words: pd.DataFrame, 

2290 fixations: pd.DataFrame, 

2291 raw_gaze: pd.DataFrame | None = None, 

2292 screens: tuple[str, ...] | None = None, 

2293) -> tuple[int, int]: 

2294 """``(trials, units)`` a bundle of ``combos`` exports: with ``screens`` 

2295 chosen, only the trials that show one of them count.""" 

2296 if screens is None: 

2297 return len(combos), count_export_units(combos, words, fixations, raw_gaze) 

2298 units = export_units(combos, words, fixations, raw_gaze, screens) 

2299 if units.empty: 

2300 return 0, 0 

2301 trials = units[["participant_id", "trial_id"]].drop_duplicates() 

2302 return len(trials), len(units) 

2303 

2304 

2305def _keep_unit_trials(combos: pd.DataFrame, units: pd.DataFrame) -> pd.DataFrame: 

2306 """``combos`` cut to the trials ``units`` still holds — after a screen 

2307 choice, the trials the bundle actually exports.""" 

2308 if units.empty: 

2309 return combos.iloc[0:0] 

2310 keys = set( 

2311 zip( 

2312 units["participant_id"].astype(str), 

2313 units["trial_id"].astype(str), 

2314 strict=True, 

2315 ) 

2316 ) 

2317 held = [ 

2318 (str(pid), str(tid)) in keys 

2319 for pid, tid in zip(combos["participant_id"], combos["trial_id"], strict=True) 

2320 ] 

2321 return combos[held] 

2322 

2323 

2324def plan_from_counts(trials: int, units: int, options: ExportOptions) -> ExportPlan: 

2325 """:class:`ExportPlan` for known trial and unit counts — the formats' 

2326 multiplication, apart from the (slower) unit expansion.""" 

2327 return ExportPlan( 

2328 trials=trials, 

2329 units=units, 

2330 figure_files=units * len(options.figure_formats()), 

2331 layer_files=units * len(SCANPATH_LAYER_ORDER) * len(options.layer_formats()), 

2332 ) 

2333 

2334 

2335def describe_plan(plan: ExportPlan) -> str: 

2336 """One line for the panel: what Build export is about to write.""" 

2337 

2338 def plural(n: int, word: str) -> str: 

2339 return f"{n:,} {word}{'' if n == 1 else 's'}" 

2340 

2341 readings = plural(plan.trials, "trial") 

2342 if plan.units != plan.trials: 

2343 readings += f" ({plural(plan.units, 'screen')})" 

2344 parts = [] 

2345 if plan.figure_files: 

2346 parts.append(plural(plan.figure_files, "figure file")) 

2347 if plan.layer_files: 

2348 parts.append(f"up to {plural(plan.layer_files, 'layer file')}") 

2349 if not parts: 

2350 return f"Exports {readings}." 

2351 return f"Exports {readings}: {' and '.join(parts)}." 

2352 

2353 

2354def _package_version() -> str: 

2355 from scanpath_studio import __version__ 

2356 

2357 return __version__ 

2358 

2359 

2360def _plural(n: int, word: str) -> str: 

2361 return f"{n:,} {word}{'' if n == 1 else 's'}" 

2362 

2363 

2364def _scope_lines( 

2365 options: ExportOptions, combos: pd.DataFrame, units: pd.DataFrame 

2366) -> list[str]: 

2367 """The README's *Scope* section: which trials the bundle was built from.""" 

2368 if options.scope == "trial": 

2369 chosen = ( 

2370 f"one trial (participant {options.scope_participant}, " 

2371 f"trial {options.scope_trial})" 

2372 ) 

2373 elif options.scope == "participant": 

2374 chosen = f"one participant ({options.scope_participant})" 

2375 elif options.scope == "text": 

2376 chosen = f"one text ({options.scope_text})" 

2377 elif options.export_unfiltered: 

2378 chosen = "the whole dataset, ignoring the trial filters" 

2379 else: 

2380 chosen = "the trials passing the trial filters" 

2381 lines = [] 

2382 if options.dataset_name: 

2383 lines.append(f"- Dataset: {options.dataset_name}") 

2384 lines += [ 

2385 f"- Trials: {chosen}", 

2386 f"- {_plural(len(combos), 'trial')}" 

2387 + (f" ({_plural(len(units), 'screen')})" if len(units) != len(combos) else ""), 

2388 ] 

2389 if options.screens is not None: 

2390 lines.append(f"- Screens: only {', '.join(options.screens)}") 

2391 return lines 

2392 

2393 

2394def bulk_export( 

2395 combos: pd.DataFrame, 

2396 words: pd.DataFrame, 

2397 fixations: pd.DataFrame, 

2398 *, 

2399 canvas_width: int, 

2400 canvas_height: int, 

2401 base_font_size: int, 

2402 font_family: str, 

2403 x_field: str, 

2404 y_field: str, 

2405 settings: dict, 

2406 options: ExportOptions, 

2407 raw_gaze: pd.DataFrame | None = None, 

2408 progress_callback=None, 

2409 status_callback: StatusCallback | None = None, 

2410 metadata_rows_for=None, 

2411 annotation_records: list[dict] | None = None, 

2412 annotation_dataset: str | None = None, 

2413 column_names: Mapping[str, ColumnNames] | None = None, 

2414) -> tuple[bytes, ExportProgress]: 

2415 """Build a zip archive of selected artifacts and return its bytes. 

2416 

2417 ``column_names`` (DATA-66) is the dataset's map per table 

2418 (``{"fixations": …, "words": …, "raw_gaze": …}``): each table is written 

2419 with the columns its file named under those names, the README's data 

2420 dictionary says where every column came from, and a ``columns.json`` 

2421 records the map; patterns accept either name. Without it (headless 

2422 callers, until the API carries maps) the tables keep the internal names. 

2423 

2424 ``metadata_rows_for(participant, trial, text_id)`` (EXP-22) returns a 

2425 trial's metadata-table rows for ``{table.field}`` patterns — the app passes 

2426 ``metadata.pattern_rows``; headless callers have no attached tables. 

2427 

2428 ``annotation_records`` (UX-179) are the annotations to write as 

2429 ``annotations.json`` when ``options.include_annotations`` is set, in 

2430 ``annotations.current_records()``'s shape; only those on the exported 

2431 trials go in. The app passes the session's; headless callers have none. 

2432 ``annotation_dataset`` is the dataset they were made on, which the file 

2433 names (annotations file schema 3, DATA-48). 

2434 

2435 progress_callback (if given) is invoked with an ExportProgress after every 

2436 trial so the UI can update a progress bar. 

2437 

2438 Raises ``ValueError`` before any work when ``options.path_pattern`` is not 

2439 a path that stays inside the ZIP (:func:`path_structure_error`). 

2440 """ 

2441 pattern_problem = path_structure_error(options.path_pattern) 

2442 if pattern_problem: 

2443 raise ValueError(pattern_problem) 

2444 combos = _apply_scope(combos, options) 

2445 units = export_units(combos, words, fixations, raw_gaze, options.screens) 

2446 if options.screens is not None: 

2447 # Only the trials that show a chosen screen are exported — for the 

2448 # annotations, the README and every per-trial table alike. 

2449 combos = _keep_unit_trials(combos, units) 

2450 # UX-179's annotations, cut to the exported trials once: the README says 

2451 # the file is there exactly when the writer below writes it (round 10). 

2452 annotations_kept: list[dict] = [] 

2453 if options.include_annotations and annotation_records: 

2454 from .annotations import records_in, records_to_store 

2455 

2456 trials = zip(combos["participant_id"], combos["trial_id"], strict=True) 

2457 annotations_kept = records_in(records_to_store(annotation_records), trials) 

2458 maps = { 

2459 table: names 

2460 for table, names in (column_names or {}).items() 

2461 if names is not None and names.entries 

2462 } 

2463 every_table = across_tables(maps) 

2464 # Each table's columns as written, decided once from the whole table so 

2465 # every per-trial file of it has the same columns — and what `columns.json` 

2466 # and the README then describe. 

2467 whole = { 

2468 table: shareable_frame(frame) 

2469 for table, frame in ( 

2470 ("fixations", fixations), 

2471 ("words", words), 

2472 ("raw_gaze", raw_gaze), 

2473 ) 

2474 if table in maps and frame is not None and not frame.empty 

2475 } 

2476 hidden = { 

2477 table: maps[table].redundant_aliases(frame) for table, frame in whole.items() 

2478 } 

2479 

2480 def names_for(artifact: str) -> tuple[ColumnNames, set[str] | None]: 

2481 """The map an artifact is written with, and the aliases it leaves out: 

2482 its own table's for the three tables, only the ids for a derived one.""" 

2483 table = _ARTIFACT_TABLE.get(artifact) 

2484 if table in maps: 

2485 return maps[table], hidden.get(table) 

2486 return every_table.identity(), None 

2487 

2488 progress = ExportProgress(total_trials=len(units)) 

2489 started = perf_counter() 

2490 emit_status( 

2491 status_callback, 

2492 ExportStage.PREPARING, 

2493 "Preparing trials and export manifest…", 

2494 started_at=started, 

2495 ) 

2496 buf = io.BytesIO() 

2497 zf = zipfile.ZipFile(buf, "w", compression=zipfile.ZIP_DEFLATED) 

2498 # The bundle's inventory, written last as `index.csv`: every file in it at 

2499 # its actual path, and every requested file that failed, with the reading 

2500 # and screen it belongs to. 

2501 inventory: list[dict] = [] 

2502 

2503 def _inventory(path, artifact, status, unit=None, note="", fmt=None): 

2504 inventory.append( 

2505 { 

2506 "path": path, 

2507 "artifact": artifact, 

2508 "format": PurePosixPath(path).suffix.lstrip(".") 

2509 if fmt is None 

2510 else fmt, 

2511 "participant_id": (unit or {}).get("participant_id", ""), 

2512 "trial_id": (unit or {}).get("trial_id", ""), 

2513 "screen_id": (unit or {}).get("screen_id", ""), 

2514 "status": status, 

2515 "note": note, 

2516 } 

2517 ) 

2518 

2519 # EXP-23: each table's per-trial frames, stacked into `aggregate/all_<table>` 

2520 # when the tables are combined. The family's words + fixations are kept 

2521 # either way, for the reader summary, which only exists across trials. 

2522 combined: dict[str, list[pd.DataFrame]] = {} 

2523 family_words: list[pd.DataFrame] = [] 

2524 family_fixations: list[pd.DataFrame] = [] 

2525 # AN-32 / EXP-23: the export writes the reading measures the dataset 

2526 # brought and computes none — so with none, the word tables are left out 

2527 # and the README says why. 

2528 measures_wanted = options.include_measures or options.include_analysis_family 

2529 brought = brought_reading_measures(words) 

2530 # What the bundle's own tables are written as, for its README and its 

2531 # `columns.json`: only the tables it writes. 

2532 exported = { 

2533 "fixations": options.include_fixations or options.include_analysis_family, 

2534 "words": measures_wanted and bool(brought), 

2535 "raw_gaze": options.include_raw_gaze, 

2536 } 

2537 written = { 

2538 table: written_columns(frame, maps[table]) 

2539 for table, frame in whole.items() 

2540 if exported[table] 

2541 } 

2542 measure_headers = { 

2543 row["canonical"]: row["column"] for row in written.get("words", []) 

2544 } 

2545 

2546 readme_lines = [ 

2547 "# Bulk export", 

2548 f"Generated: {_local_stamp()}", 

2549 "", 

2550 f"Made with: {CITATION['title']}", 

2551 f"Tool authors: {CITATION['authors']}", 

2552 f"Version: {_package_version()}", 

2553 f"DOI: https://doi.org/{CITATION['doi']}", 

2554 "", 

2555 "## Scope", 

2556 *_scope_lines(options, combos, units), 

2557 "", 

2558 "## Layout", 

2559 "- `index.csv` lists every file in this bundle — its path, what it is, " 

2560 "its participant, trial and screen — and every requested file that " 

2561 "failed (`status` = `failed`) or reading skipped (`skipped`). A file " 

2562 "type that was not requested is not listed.", 

2563 "- `per_trial/<participant>__<trial>/` holds each trial's files, " 

2564 "unless the File naming pattern moved them (`index.csv` has every path).", 

2565 "- A trial shown on several screens adds `screens/screen-001-<id>/` " 

2566 "inside its folder.", 

2567 *( 

2568 [ 

2569 "- `aggregate/` holds each table once, every trial in this run " 

2570 "stacked in it (*Combine all trials into one file*)." 

2571 ] 

2572 if options.combine_trials and options.any_table() 

2573 else ( 

2574 ["- `aggregate/` holds the reader summary, across every trial."] 

2575 if options.include_analysis_family 

2576 else [] 

2577 ) 

2578 ), 

2579 *( 

2580 [ 

2581 "- `annotations.json` holds the favorites, tags and notes on " 

2582 "these trials; import it on the app's Data Management page → Annotations." 

2583 ] 

2584 if annotations_kept 

2585 else [] 

2586 ), 

2587 "", 

2588 "## Data dictionary", 

2589 *( 

2590 [ 

2591 "Columns keep the names they have in the dataset's own files. " 

2592 "A column Scanpath Studio built, converted or computed keeps its " 

2593 "internal name; `columns.json` maps every column below to its " 

2594 "internal name, for scripts that work across datasets. Columns " 

2595 "not listed are written under their own names. A table the app " 

2596 "derives (a summary, the saccades) carries the dataset's ids under " 

2597 "these names, and its own values under the app's.", 

2598 *dictionary_lines(written, maps), 

2599 *( 

2600 ["", "Reading measures in the word tables:"] 

2601 if measures_wanted 

2602 else [] 

2603 ), 

2604 ] 

2605 if any(written.values()) 

2606 else [ 

2607 "Column names (Scanpath Studio's standard names):", 

2608 "- participant_id, trial_id, text_id, word_id", 

2609 "- screen_id, screen_index (multipart trials only)", 

2610 "- x, y, width, height (word bounding boxes in screen px)", 

2611 "- x, y, duration_ms, timestamp_ms (fixations)", 

2612 ] 

2613 ), 

2614 *( 

2615 [f"- {measure_headers.get(column, column)}" for column in brought] 

2616 if brought 

2617 else ["- (no reading measures — see below)"] 

2618 if measures_wanted 

2619 else [] 

2620 ), 

2621 "", 

2622 "## Reading measures", 

2623 "The word tables carry the reading measures the dataset brought, as " 

2624 "mapped on the app's Data Management page; Scanpath Studio computes none of them.", 

2625 *( 

2626 [ 

2627 "", 

2628 "This dataset brought none, so the bundle has no word-measure " 

2629 "table. Map them in the app on Data Management → Edit dataset → " 

2630 "Reading measures.", 

2631 ] 

2632 if measures_wanted and not brought 

2633 else [] 

2634 ), 

2635 *( 

2636 ["", f"Demo data note: {CITATION['corpus_note']}"] 

2637 if options.dataset_name == DEMO_CHOICE 

2638 else [] 

2639 ), 

2640 ] 

2641 zf.writestr("README.md", "\n".join(readme_lines)) 

2642 _inventory("README.md", "readme", "written") 

2643 if any(written.values()): 

2644 zf.writestr("columns.json", json.dumps(columns_manifest(written), indent=2)) 

2645 _inventory("columns.json", "columns", "written") 

2646 if options.include_analysis_family: 

2647 zf.writestr( 

2648 "run_config.json", 

2649 json.dumps( 

2650 { 

2651 "generated_at": datetime.now(UTC).isoformat(), 

2652 "settings": settings, 

2653 "preprocessing": settings.get("preprocessing", {}), 

2654 }, 

2655 indent=2, 

2656 default=str, 

2657 ), 

2658 ) 

2659 _inventory("run_config.json", "run_config", "written") 

2660 

2661 # One warm Kaleido browser for every trial's figure (see _figure_renderer) 

2662 # instead of cold-starting Chrome on each render. HTML needs no browser, so 

2663 # only spin Kaleido up when a raster/vector format was requested (combined 

2664 # figure or the per-layer breakdown). 

2665 figure_formats = options.figure_formats() 

2666 layer_formats = options.layer_formats() 

2667 # EXP-1: a user pattern can map two trials to the same path (one that omits 

2668 # the trial id, say). Two zip entries at one name silently loses a file, so 

2669 # `resolve_export_path` disambiguates against what's already been written. 

2670 used_paths: set = set() 

2671 # The readers, trials and texts the bundle actually holds — what its 

2672 # metadata tables are narrowed to at the end. 

2673 exported_pairs: set[tuple[str, str]] = set() 

2674 exported_texts: set[str] = set() 

2675 emit_status( 

2676 status_callback, 

2677 ( 

2678 ExportStage.STARTING_RENDERER 

2679 if options.needs_kaleido() 

2680 else ExportStage.ENCODING_WRITING 

2681 ), 

2682 ( 

2683 "Starting one shared Chrome/Kaleido renderer…" 

2684 if options.needs_kaleido() 

2685 else "Writing selected files…" 

2686 ), 

2687 started_at=started, 

2688 ) 

2689 with _figure_renderer(options.needs_kaleido()) as render_figure: 

2690 for combo in units.itertuples(index=False): 

2691 # The cancel checkpoint (`progress.Cancelled`), between units so a 

2692 # stopped build never leaves one half-written; a no-op headlessly. 

2693 try: 

2694 report_progress( 

2695 progress.finished_trials, progress.total_trials, unit="screens" 

2696 ) 

2697 except BaseException: # `progress.Cancelled`: close the zip, go 

2698 zf.close() 

2699 raise 

2700 participant = combo.participant_id 

2701 trial = combo.trial_id 

2702 screen_id = getattr(combo, SCREEN_ID, None) 

2703 screen_index = getattr(combo, SCREEN_INDEX, None) 

2704 screen_slug = ( 

2705 f"screen-{int(screen_index):03d}-{_safe_id(screen_id)}" 

2706 if screen_id is not None and pd.notna(screen_id) 

2707 else "" 

2708 ) 

2709 slug = f"{_safe_id(participant)}__{_safe_id(trial)}" 

2710 if screen_slug: 

2711 slug += f"__{screen_slug}" 

2712 unit_ids = { 

2713 "participant_id": str(participant), 

2714 "trial_id": str(trial), 

2715 "screen_id": str(screen_id) if screen_slug else "", 

2716 } 

2717 

2718 # Slice via the same str-normalized position index the live view uses 

2719 # (utils.extract_trial), so the export selects *exactly* what the trial 

2720 # picker shows — not a raw dtype-sensitive boolean mask that can silently 

2721 # miss rows the view finds. 

2722 trial_words = extract_trial(words, participant, trial) 

2723 trial_fix = extract_trial(fixations, participant, trial) 

2724 trial_raw_gaze = ( 

2725 extract_trial(raw_gaze, participant, trial) 

2726 if raw_gaze is not None and not raw_gaze.empty 

2727 else pd.DataFrame() 

2728 ) 

2729 if screen_slug: 

2730 trial_words = extract_part( 

2731 trial_words, participant, trial, str(screen_id) 

2732 ) 

2733 trial_fix = extract_part(trial_fix, participant, trial, str(screen_id)) 

2734 trial_raw_gaze = ( 

2735 extract_part(trial_raw_gaze, participant, trial, str(screen_id)) 

2736 if not trial_raw_gaze.empty and SCREEN_ID in trial_raw_gaze.columns 

2737 else pd.DataFrame() 

2738 ) 

2739 

2740 # Skip only a genuinely empty trial (nothing to draw). The figure 

2741 # builder renders from fixations alone (words optional — boxes/labels) 

2742 # or from words alone (AOI layout), and the live view does too; so a 

2743 # words-that-don't-join / fixations-only trial must still export 

2744 # instead of being skipped with "empty data" (VIZ-5). VIZ-45: and 

2745 # a trial recorded as raw gaze alone is drawn from its samples. 

2746 if trial_words.empty and trial_fix.empty and trial_raw_gaze.empty: 

2747 progress.finished_trials += 1 

2748 progress.trials_skipped += 1 

2749 progress.errors.append(f"{slug}: empty data, skipped") 

2750 _inventory("", "reading", "skipped", unit_ids, "no data", fmt="") 

2751 if progress_callback: 

2752 progress_callback(progress) 

2753 continue 

2754 

2755 exported_pairs.add((str(participant), str(trial))) 

2756 exported_texts.update( 

2757 _unit_text_ids(combo._asdict(), trial_words, trial_fix, trial_raw_gaze) 

2758 ) 

2759 

2760 # A screen's own canvas, for its figure and its plot config alike. 

2761 unit_canvas = ( 

2762 screen_canvas_size(trial_words) 

2763 or screen_canvas_size(trial_fix) 

2764 or screen_canvas_size(trial_raw_gaze) 

2765 ) 

2766 unit_width = int(unit_canvas[0] if unit_canvas else canvas_width) 

2767 unit_height = int(unit_canvas[1] if unit_canvas else canvas_height) 

2768 

2769 # EXP-1/EXP-2: everything this trial's paths, title and caption can 

2770 # substitute — its combo row plus counts and the settings summary. 

2771 fields = pattern_fields( 

2772 participant, 

2773 trial, 

2774 trial_words, 

2775 trial_fix, 

2776 settings, 

2777 combo_row=combo._asdict(), 

2778 dataset_name=options.dataset_name, 

2779 column_names=every_table, 

2780 metadata_rows=( 

2781 metadata_rows_for( 

2782 participant, trial, combo._asdict().get("text_id") 

2783 ) 

2784 if metadata_rows_for is not None 

2785 else None 

2786 ), 

2787 ) 

2788 

2789 def _path( 

2790 artifact: str, ext: str, _f=fields, _slug=screen_slug, reserve=True 

2791 ) -> str: 

2792 # `reserve=False` names a file that failed: the path it would 

2793 # have had, for the inventory, without taking it from the next. 

2794 if _slug: 

2795 artifact = f"screens/{_slug}/{artifact}" 

2796 return resolve_export_path( 

2797 options.path_pattern, 

2798 _f, 

2799 artifact=artifact, 

2800 ext=ext, 

2801 used=used_paths if reserve else set(used_paths), 

2802 ) 

2803 

2804 title = ( 

2805 render_pattern(options.title_pattern, fields) 

2806 if options.title_pattern 

2807 else "" 

2808 ) 

2809 caption = ( 

2810 render_pattern(options.caption_pattern, fields) 

2811 if options.caption_pattern 

2812 else "" 

2813 ) 

2814 

2815 if options.needs_figure(): 

2816 emit_status( 

2817 status_callback, 

2818 ExportStage.RASTERIZING, 

2819 f"Rendering {progress.finished_trials + 1} of {progress.total_trials}…", 

2820 started_at=started, 

2821 completed=progress.finished_trials, 

2822 total=progress.total_trials, 

2823 ) 

2824 fig = None 

2825 try: 

2826 # EXP-4 / VIZ-24: apply the PRE-3 drift correction to the 

2827 # figure's fixations (a no-op when "Off"), so the batch 

2828 # matches the corrected figure on screen. `trial_fix` itself 

2829 # stays uncorrected — the tables below export the recording. 

2830 fig_fix, connector_y = _drift_corrected_for_figure( 

2831 trial_fix, trial_words, settings 

2832 ) 

2833 render_values = { 

2834 name: settings[name] 

2835 for name in STATIC_FIGURE_OPTIONS 

2836 if name in settings 

2837 } 

2838 render_settings = FigureSettings.from_mapping( 

2839 render_values, 

2840 canvas_width=unit_width, 

2841 canvas_height=unit_height, 

2842 base_font_size=int(base_font_size), 

2843 font_family=font_family, 

2844 x_field=x_field, 

2845 y_field=y_field, 

2846 show_connectors=connector_y is not None, 

2847 connector_y=connector_y, 

2848 # A corrected figure colours by line, exactly as the 

2849 # on-screen static path forces it (tabs.py PRE-3 

2850 # overrides) — else the batch differs in colouring. 

2851 color_by_line=bool(settings.get("color_by_line", False)) 

2852 or fig_fix is not trial_fix, 

2853 ) 

2854 # VIZ-45: the trial's own samples, as the live figure 

2855 # gets them (the layer self-gates on `show_raw_gaze`). 

2856 fig = make_scanpath_figure( 

2857 trial_words, 

2858 fig_fix, 

2859 settings=render_settings, 

2860 raw_gaze=trial_raw_gaze if not trial_raw_gaze.empty else None, 

2861 ) 

2862 # EXP-2: stamp the title/caption BEFORE measuring the output 

2863 # size — the bands grow the figure, and rendering at the 

2864 # pre-title size would crop them off. 

2865 annotate_figure(fig, title=title, caption=caption) 

2866 except Exception as exc: 

2867 fig = None 

2868 # Its formats, and its layer set when one was asked for. 

2869 progress.figures_failed += len(figure_formats) + bool(layer_formats) 

2870 progress.errors.append(f"{slug}: figure export failed ({exc})") 

2871 for fmt in figure_formats: 

2872 _inventory( 

2873 _path("figure", fmt, reserve=False), 

2874 "figure", 

2875 "failed", 

2876 unit_ids, 

2877 str(exc), 

2878 ) 

2879 if layer_formats: 

2880 _inventory("", "layers", "failed", unit_ids, str(exc), fmt="") 

2881 # EXP-24: each format on its own, so a missing browser costs the 

2882 # PNG/SVG/PDF and still leaves the trial's HTML in the zip. 

2883 if fig is not None: 

2884 # Render at the figure's own fitted size (not the raw 

2885 # monitor canvas) so the exported reading text matches the 

2886 # on-screen scale. 

2887 out_w = int(fig.layout.width or unit_width) 

2888 out_h = int(fig.layout.height or unit_height) 

2889 for fmt in figure_formats: 

2890 try: 

2891 if fmt == "html": 

2892 # Browser-free + interactive; no Kaleido needed. 

2893 data = fig.to_html( 

2894 include_plotlyjs=html_plotlyjs( 

2895 options.html_self_contained 

2896 ), 

2897 full_html=True, 

2898 config={**PLOTLY_CONFIG}, 

2899 ).encode("utf-8") 

2900 else: 

2901 scale = options.png_scale if fmt == "png" else 1 

2902 data = render_figure(fig, fmt, out_w, out_h, scale) 

2903 path = _path("figure", fmt) 

2904 zf.writestr(path, data) 

2905 except Exception as exc: 

2906 progress.figures_failed += 1 

2907 progress.errors.append( 

2908 f"{slug}: {fmt.upper()} figure export failed ({exc})" 

2909 ) 

2910 _inventory( 

2911 _path("figure", fmt, reserve=False), 

2912 "figure", 

2913 "failed", 

2914 unit_ids, 

2915 str(exc), 

2916 ) 

2917 continue 

2918 _inventory(path, "figure", "written", unit_ids) 

2919 progress.bytes_written += len(data) 

2920 progress.figures_written += 1 

2921 

2922 # VIZ-5: per-layer breakdown into `layers/<layer>.<fmt>` — each a 

2923 # copy of the figure with only one layer's elements, same 

2924 # size/ranges so they register when stacked in Illustrator. Kept in 

2925 # its own try so a layer-render failure is reported as such and 

2926 # doesn't get misattributed to the combined figure (which may have 

2927 # already been written above). 

2928 if layer_formats and fig is not None: 

2929 try: 

2930 out_w = int(fig.layout.width or unit_width) 

2931 out_h = int(fig.layout.height or unit_height) 

2932 for layer_name, layer_fig in split_scanpath_layers(fig).items(): 

2933 for fmt in layer_formats: 

2934 scale = options.png_scale if fmt == "png" else 1 

2935 data = render_figure( 

2936 layer_fig, fmt, out_w, out_h, scale 

2937 ) 

2938 path = _path(f"layers/{layer_name}", fmt) 

2939 zf.writestr(path, data) 

2940 _inventory( 

2941 path, f"layer:{layer_name}", "written", unit_ids 

2942 ) 

2943 progress.bytes_written += len(data) 

2944 progress.figures_written += 1 

2945 except Exception as exc: 

2946 progress.figures_failed += 1 

2947 progress.errors.append(f"{slug}: layer export failed ({exc})") 

2948 _inventory("", "layers", "failed", unit_ids, str(exc), fmt="") 

2949 

2950 if options.include_plot_config: 

2951 # Same guard as _drift_corrected_for_figure: correction runs 

2952 # when an algorithm is set and the trial has both frames. 

2953 algorithm = settings.get("align_algorithm") 

2954 drift_applied = ( 

2955 bool(algorithm) 

2956 and str(algorithm) != "Off" 

2957 and not trial_fix.empty 

2958 and not trial_words.empty 

2959 ) 

2960 cfg = _plot_config_dict( 

2961 participant, 

2962 trial, 

2963 unit_width, 

2964 unit_height, 

2965 x_field, 

2966 y_field, 

2967 settings, 

2968 screen_id=str(screen_id) if screen_slug else None, 

2969 drift_applied=drift_applied, 

2970 ) 

2971 # EXP-2: the title/caption are part of how the figure looked, so 

2972 # the manifest records them verbatim alongside the settings. 

2973 cfg["figure_text"] = {"title": title, "caption": caption} 

2974 # VIZ-50: the trial's samples, and whether they were recorded — 

2975 # the bundled demo's are synthesized, which its figure and its 

2976 # raw-gaze table cannot say for themselves. 

2977 if not trial_raw_gaze.empty: 

2978 cfg["raw_gaze"] = { 

2979 "points": len(trial_raw_gaze), 

2980 "synthesized": bool(settings.get("raw_gaze_synthesized")), 

2981 } 

2982 data = json.dumps(cfg, indent=2).encode("utf-8") 

2983 path = _path("plot_config", "json") 

2984 zf.writestr(path, data) 

2985 _inventory(path, "plot_config", "written", unit_ids) 

2986 progress.bytes_written += len(data) 

2987 

2988 # AN-32 / EXP-23: the word table *is* the measures table — it carries 

2989 # what the dataset brought, and nothing is computed. A trial with no 

2990 # words, or a dataset that brought no measures, writes none. 

2991 per_trial_measures = ( 

2992 trial_words 

2993 if measures_wanted and brought and not trial_words.empty 

2994 else None 

2995 ) 

2996 family = {} 

2997 if options.include_analysis_family: 

2998 measured = trial_words 

2999 analysis_fix = ( 

3000 enrich_fixations( 

3001 assign_fixations_to_words(trial_fix, trial_words), trial_words 

3002 ) 

3003 if not trial_fix.empty and not trial_words.empty 

3004 else trial_fix 

3005 ) 

3006 family = { 

3007 "fixations": analysis_fix, 

3008 "word_measures": per_trial_measures, 

3009 "saccades": saccade_table( 

3010 analysis_fix, 

3011 pixels_per_degree=settings.get("pixels_per_degree"), 

3012 raw_gaze=trial_raw_gaze, 

3013 words=trial_words, 

3014 ), 

3015 "sentence_measures": sentence_measures(measured, analysis_fix), 

3016 "trial_summary": trial_summary_table(measured, analysis_fix), 

3017 "characters": character_grid(trial_words), 

3018 "cleaning_qa": cleaning_report( 

3019 analysis_fix, 

3020 short_policy=(settings.get("preprocessing") or {}).get( 

3021 "short_policy", "Off" 

3022 ), 

3023 ), 

3024 } 

3025 

3026 # The family's word-enriched fixations and its word_measures stand 

3027 # in for the plain Fixations / Word measures files (EXP-15: two 

3028 # members with one name, and a reader keeps only one of them). 

3029 tables = { 

3030 "fixations": trial_fix 

3031 if options.include_fixations and not options.include_analysis_family 

3032 else None, 

3033 "raw_gaze": trial_raw_gaze if options.include_raw_gaze else None, 

3034 "measures": per_trial_measures 

3035 if options.include_measures and not options.include_analysis_family 

3036 else None, 

3037 **family, 

3038 } 

3039 tables = { 

3040 artifact: table 

3041 for artifact, table in tables.items() 

3042 if table is not None and not table.empty 

3043 } 

3044 if options.combine_trials: 

3045 for artifact, table in tables.items(): 

3046 combined.setdefault(artifact, []).append(table) 

3047 else: 

3048 for fmt in options.table_formats(): 

3049 for artifact, table in tables.items(): 

3050 path = _path(artifact, fmt) 

3051 progress.bytes_written += _write_table( 

3052 zf, path, table, fmt, *names_for(artifact) 

3053 ) 

3054 _inventory(path, artifact, "written", unit_ids) 

3055 if options.include_analysis_family: 

3056 family_words.append(measured) 

3057 family_fixations.append(family["fixations"]) 

3058 

3059 progress.finished_trials += 1 

3060 if progress_callback: 

3061 progress_callback(progress) 

3062 emit_status( 

3063 status_callback, 

3064 ExportStage.ENCODING_WRITING, 

3065 f"Wrote {progress.finished_trials} of {progress.total_trials}…", 

3066 started_at=started, 

3067 completed=progress.finished_trials, 

3068 total=progress.total_trials, 

3069 ) 

3070 

3071 # A reader summary has no per-trial form, so it is written across the 

3072 # exported trials whether or not the tables are combined. 

3073 if options.include_analysis_family and family_words: 

3074 summary = reader_summary_table( 

3075 pd.concat(family_words, ignore_index=True), 

3076 pd.concat(family_fixations, ignore_index=True), 

3077 ) 

3078 if not summary.empty: 

3079 combined["reader_summary"] = [summary] 

3080 stacked = { 

3081 artifact: pd.concat(frames, ignore_index=True) 

3082 for artifact, frames in combined.items() 

3083 } 

3084 for fmt in options.table_formats(): 

3085 for artifact, table in stacked.items(): 

3086 path = f"aggregate/all_{artifact}.{fmt}" 

3087 progress.bytes_written += _write_table( 

3088 zf, path, table, fmt, *names_for(artifact) 

3089 ) 

3090 _inventory(path, artifact, "written") 

3091 # DATA-20: the participant table travels as its own per-grain table rather 

3092 # than as columns smeared across the trial files — which is what keeps a 

3093 # reader attribute distinguishable from a per-fixation measurement on the 

3094 # way out, exactly as it is on the way in. Whenever one is attached, narrowed 

3095 # to `options.metadata_fields` (milestone 10's per-field opt-out; `None` is 

3096 # every field). The frame is handed over out-of-band (the settings dict 

3097 # carries only its fingerprint, so the bulk-export cache signature stays 

3098 # honest — see the note in `tabs._render_export_panel`). API/CLI callers can 

3099 # pass the frame in `settings` directly, which still works. 

3100 participant_metadata = (settings or {}).get("participant_metadata") 

3101 if participant_metadata is None: 

3102 participant_metadata = _session_participant_metadata() 

3103 participant_metadata = _rows_in_scope( 

3104 _selected_metadata_columns(participant_metadata, options.metadata_fields), 

3105 pairs=exported_pairs, 

3106 texts=exported_texts, 

3107 grain="participant", 

3108 ) 

3109 if participant_metadata is not None and not participant_metadata.empty: 

3110 for fmt in options.table_formats(): 

3111 path = f"metadata/participants.{fmt}" 

3112 progress.bytes_written += _write_table(zf, path, participant_metadata, fmt) 

3113 _inventory(path, "participant_metadata", "written") 

3114 # DATA-29: and the trial table beside it, on the same terms — its own 

3115 # per-grain file, keyed as it was attached, so a reading's attributes stay 

3116 # distinguishable from the per-fixation measurements of that reading. 

3117 trial_metadata = (settings or {}).get("trial_metadata") 

3118 if trial_metadata is None: 

3119 trial_metadata = _session_trial_metadata() 

3120 trial_metadata = _rows_in_scope( 

3121 _selected_trial_metadata_columns(trial_metadata, options.trial_metadata_fields), 

3122 pairs=exported_pairs, 

3123 texts=exported_texts, 

3124 grain="trial", 

3125 ) 

3126 if trial_metadata is not None and not trial_metadata.empty: 

3127 for fmt in options.table_formats(): 

3128 path = f"metadata/trials.{fmt}" 

3129 progress.bytes_written += _write_table(zf, path, trial_metadata, fmt) 

3130 _inventory(path, "trial_metadata", "written") 

3131 # And the text table, the third grain — same reasoning again. 

3132 text_metadata = (settings or {}).get("text_metadata") 

3133 if text_metadata is None: 

3134 text_metadata = _session_text_metadata() 

3135 text_metadata = _rows_in_scope( 

3136 _selected_text_metadata_columns(text_metadata, options.text_metadata_fields), 

3137 pairs=exported_pairs, 

3138 texts=exported_texts, 

3139 grain="text", 

3140 ) 

3141 if text_metadata is not None and not text_metadata.empty: 

3142 for fmt in options.table_formats(): 

3143 path = f"metadata/texts.{fmt}" 

3144 progress.bytes_written += _write_table(zf, path, text_metadata, fmt) 

3145 _inventory(path, "text_metadata", "written") 

3146 # UX-179: the exported trials' annotations, in the Data → Annotations file 

3147 # format, so the bundle's notes can be imported back into the app. 

3148 if annotations_kept: 

3149 from .annotations import records_to_store, serialize 

3150 

3151 data = serialize( 

3152 records_to_store(annotations_kept), dataset=annotation_dataset 

3153 ).encode("utf-8") 

3154 zf.writestr("annotations.json", data) 

3155 _inventory("annotations.json", "annotations", "written") 

3156 progress.bytes_written += len(data) 

3157 emit_status( 

3158 status_callback, 

3159 ExportStage.FINALIZING, 

3160 "Finalizing and compressing the zip archive…", 

3161 started_at=started, 

3162 completed=progress.finished_trials, 

3163 total=progress.total_trials, 

3164 ) 

3165 _write_inventory(zf, inventory) 

3166 progress.files_written = len(zf.namelist()) 

3167 zf.close() 

3168 buf.seek(0) 

3169 result = buf.getvalue() 

3170 emit_status( 

3171 status_callback, 

3172 ExportStage.READY, 

3173 "Export archive is ready.", 

3174 started_at=started, 

3175 completed=progress.finished_trials, 

3176 total=progress.total_trials, 

3177 ) 

3178 return result, progress