Coverage for scanpath_studio/export.py: 94%
1024 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Configurable bulk export of figures and tabular data for filtered trials.
3This module powers the "Bulk export" button. Users pick which artifacts they
4want per trial (PNG, SVG, JSON plot config, fixations CSV/Parquet, per-word
5measures CSV/Parquet), or each table once with every trial stacked in it
6(EXP-23's *Combine all trials into one file*). Everything is packaged into a
7single zip archive with a clean folder structure:
9 bulk_export_<timestamp>.zip
10 ├─ per_trial/
11 │ ├─ <participant>__<trial>/
12 │ │ ├─ figure.png
13 │ │ ├─ figure.svg
14 │ │ ├─ layers/ (VIZ-5, optional)
15 │ │ │ ├─ word_boxes.svg
16 │ │ │ ├─ fixations.svg
17 │ │ │ ├─ saccades.svg
18 │ │ │ └─ … (one per visible layer)
19 │ │ ├─ plot_config.json
20 │ │ ├─ fixations.csv (and/or .parquet)
21 │ │ └─ measures.csv (and/or .parquet)
22 │ ├─ ...
23 ├─ aggregate/ (combined tables, instead of the per-trial ones)
24 │ ├─ all_fixations.csv (and/or .parquet)
25 │ └─ all_measures.csv (and/or .parquet)
26 └─ annotations.json (UX-179, optional)
27"""
29from __future__ import annotations
31import io
32import json
33import re
34import zipfile
35from collections.abc import Mapping
36from contextlib import contextmanager, nullcontext
37from dataclasses import dataclass, field
38from datetime import UTC, datetime
39from pathlib import PurePosixPath
40from time import perf_counter
42import pandas as pd
44from .aggregation import reader_summary_table, trial_summary_table
45from .column_names import (
46 ColumnNames,
47 across_tables,
48 as_written,
49 columns_manifest,
50 dictionary_lines,
51 written_columns,
52)
53from .constants import (
54 CITATION,
55 DEFAULT_FIXATION_COLOR,
56 DEFAULT_FIXATION_SYMBOL,
57 DEFAULT_LINE_SPACING,
58 DEFAULT_PALETTE,
59 DEMO_CHOICE,
60 ICONS,
61 PLOTLY_CONFIG,
62 SACCADE_CLASS_ORDER,
63 UNIFORM_COLOR_FIELD,
64 computed_measures_enabled,
65 drift_correction_enabled,
66)
67from .data import brought_reading_measures, shareable_frame
68from .export_status import ExportStage, StatusCallback, emit_status
69from .fields import panel_field
70from .measures import assign_fixations_to_words, enrich_fixations
71from .multipart import (
72 SCREEN_ID,
73 SCREEN_INDEX,
74 extract_part,
75 part_catalog,
76 screen_canvas_size,
77)
78from .plots import (
79 SCANPATH_LAYER_ORDER,
80 STATIC_FIGURE_OPTIONS,
81 FigureSettings,
82 _plotly_literal,
83 make_scanpath_figure,
84 normalize_legend_layout,
85 split_scanpath_layers,
86)
87from .preprocessing import (
88 character_grid,
89 cleaning_report,
90 saccade_table,
91 sentence_measures,
92)
93from .progress import report as report_progress
94from .utils import extract_trial
97def _local_stamp() -> str:
98 """Now in this computer's local time, with its UTC offset said (#374, F28):
99 ``2026-10-06 21:21:01 (UTC+03:00)``."""
100 now = datetime.now().astimezone()
101 offset = now.strftime("%z")
102 return f"{now:%Y-%m-%d %H:%M:%S} (UTC{offset[:3]}:{offset[3:]})"
105# --- #374 F28 · a raster figure's print size ---------------------------------
106#: The units a print width is given in → millimetres per unit.
107PRINT_UNITS = {"mm": 1.0, "in": 25.4}
108#: The resolution a print width is drawn at when none is named.
109DEFAULT_PRINT_DPI = 300
110#: #374 F28 — the print width and dpi every surface accepts: Export → Current
111#: figure's boxes, a link or settings file (clamped), `render` and `save_figure`
112#: (refused outside them).
113PRINT_WIDTH_BOUNDS = (1.0, 2000.0)
114PRINT_DPI_BOUNDS = (50, 2400)
115#: What the app's Export → Current figure PNG is drawn at without a print width
116#: (`tabs._PNG_EXPORT_SCALE`): three pixels per figure pixel.
117SCREEN_PNG_SCALE = 3
120def print_width_px(width: float, unit: str = "mm", dpi: int = DEFAULT_PRINT_DPI) -> int:
121 """The pixel width of a raster figure ``width`` ``unit`` wide at ``dpi``:
122 180 mm at 600 dpi is 4,252 px."""
123 if unit not in PRINT_UNITS:
124 raise ValueError(f"Unknown width unit {unit!r}; use 'mm' or 'in'.")
125 if not width or width <= 0 or not dpi or dpi <= 0:
126 raise ValueError("A print width and its dpi must both be positive.")
127 low, high = PRINT_WIDTH_BOUNDS
128 if not low <= float(width) <= high:
129 raise ValueError(f"A print width must be {low:g}–{high:g} {unit}.")
130 low, high = PRINT_DPI_BOUNDS
131 if not low <= float(dpi) <= high:
132 raise ValueError(f"A print dpi must be {low}–{high}.")
133 return max(1, round(float(width) * PRINT_UNITS[unit] / 25.4 * float(dpi)))
136def print_scale(
137 figure_width: int, width: float, unit: str = "mm", dpi: int = DEFAULT_PRINT_DPI
138) -> float:
139 """The ``scale`` that draws a ``figure_width``-px figure ``width`` ``unit``
140 wide at ``dpi`` (the height follows the figure's own aspect)."""
141 return print_width_px(width, unit, dpi) / float(figure_width)
144def png_save_kwargs(
145 width: float | None, unit: str = "mm", dpi: int | None = None
146) -> dict:
147 """The `api.save_figure` keywords that write the PNG Export → *Current
148 figure* writes: the print width at its dpi, else the screen size at
149 `SCREEN_PNG_SCALE` — so Share → Code writes the same pixel size."""
150 if not width:
151 return {"scale": SCREEN_PNG_SCALE}
152 return {f"width_{unit}": float(width), "dpi": int(dpi or DEFAULT_PRINT_DPI)}
155def set_png_dpi(path, dpi: int) -> None:
156 """Stamp ``dpi`` into a PNG's header (``pHYs``), so a layout program
157 places it at its print size. The pixels are untouched."""
158 from PIL import Image
160 with Image.open(path) as image:
161 image.load()
162 image.save(path, format="PNG", dpi=(dpi, dpi))
165# --- EXP-1 · customizable export paths ---------------------------------------
166# A zip of 200 trials landed with names the tool chose, which is rarely how a
167# user organizes figures for a paper. The path of every artifact is now a
168# *pattern* over the trial's own fields, so a batch can be dropped straight into
169# an existing folder structure. The default reproduces the historical layout
170# exactly, so nothing changes unless the pattern is edited.
171DEFAULT_PATH_PATTERN = "per_trial/{participant_id}__{trial_id}/{artifact}.{ext}"
172# EXP-2 · a figure pulled into a paper or a slide loses its provenance the moment
173# it leaves the app, so it can carry its own. Same substitution as the path
174# patterns; empty means "no title / no caption".
175DEFAULT_TITLE_PATTERN = "{participant_id} · {trial_id}"
176DEFAULT_CAPTION_PATTERN = "{text_id} · {n_fixations} fixations · {settings}"
179@dataclass
180class ExportOptions:
181 """User-chosen export artifacts.
183 Defaults: figures on (PNG + SVG), tabular data off. The scope fields
184 narrow the set of trials when not "all".
185 """
187 include_png: bool = True
188 include_svg: bool = True
189 include_pdf: bool = False
190 # HTML is a browser-free figure format (fig.to_html — no Kaleido/Chrome) and
191 # stays interactive; handled specially in the export loop.
192 include_html: bool = False
193 include_plot_config: bool = True
194 # UX-179: the exported trials' annotations (favorites, tags, notes) as one
195 # `annotations.json` at the bundle root — the file 🗂️ Data → Annotations
196 # exports and imports. Off by default: notes can be personal.
197 include_annotations: bool = False
198 include_fixations: bool = False
199 # VIZ-45: each trial's raw (sample-level) gaze as its own table, written as
200 # recorded — for a raw-gaze-only dataset it is the only recording there is.
201 include_raw_gaze: bool = False
202 # AN-32 / EXP-23: the word table with the reading measures the dataset
203 # *brought*. Export computes none, as the Corpus Analysis page doesn't.
204 include_measures: bool = False
205 include_analysis_family: bool = False
206 # EXP-23: write each chosen table once, every exported trial stacked, as
207 # `aggregate/all_<table>` — instead of one file per trial. Replaced the
208 # "Mega-table", which stacked two of the tables beside the per-trial ones.
209 combine_trials: bool = False
210 # VIZ-5: also drop a per-layer breakdown of the figure (word boxes / fixations
211 # / saccades / heatmap / labels / stimulus image) into `layers/` so each can be
212 # restyled independently in Illustrator / Inkscape. Uses the selected vector /
213 # raster formats (SVG when none was picked — the publication default).
214 separable_layers: bool = False
215 table_format: str = "csv" # "csv" | "parquet" | "both"
216 png_scale: int = 2
217 # EXP-1: where each artifact lands inside the zip. `{artifact}` / `{ext}` name
218 # the file (figure / fixations / measures / plot_config / layer names); every
219 # other placeholder comes from the trial. The default is the historical layout.
220 path_pattern: str = DEFAULT_PATH_PATTERN
221 # EXP-2: an optional title + caption rendered into the exported image and
222 # recorded in the per-trial manifest. Empty = off (the historical behaviour).
223 title_pattern: str = ""
224 caption_pattern: str = ""
225 # VIZ-36: what `{dataset_name}` substitutes to. The app fills it from the
226 # dataset picker's label; a headless caller that says nothing gets "", so
227 # the placeholder renders empty rather than erroring on a surface nobody is
228 # watching.
229 dataset_name: str = ""
230 # DATA-20 milestone 10: which columns of the attached participant table go
231 # into `metadata/participants.*`. `None` is every field (the default, and
232 # what a headless caller that knows nothing about this gets); a tuple is
233 # exactly those, always alongside the reader id; an **empty** tuple leaves
234 # the table out of the bundle. Reader attributes are the most re-identifying
235 # thing an export can carry, so the opt-out has to be per field rather than
236 # all-or-nothing.
237 metadata_fields: tuple[str, ...] | None = None
238 # DATA-29: the same per-field opt-out for the attached *trial* table,
239 # written to `metadata/trials.*`. Kept as its own field rather than
240 # folded into `metadata_fields` because the two tables are attached and
241 # cleared independently, and because what is safe to ship differs by
242 # grain: a reader attribute re-identifies a person, a trial attribute
243 # usually describes the material.
244 trial_metadata_fields: tuple[str, ...] | None = None
245 # And the same opt-out for the attached *text* table, written to
246 # `metadata/texts.*` — the third grain, same reasoning again.
247 text_metadata_fields: tuple[str, ...] | None = None
248 # HTML figures embed the Plotly library (opens offline, ~4.8 MB more per
249 # file) instead of loading it from cdn.plot.ly when opened. Off by default:
250 # the bundle's *Self-contained HTML* choice, shown while HTML is picked.
251 html_self_contained: bool = False
252 # When True, export operates on the whole loaded dataset, ignoring the
253 # trial-filter funnel; the caller supplies the unfiltered frames.
254 export_unfiltered: bool = False
255 scope: str = "all" # "all" | "trial" | "participant" | "text"
256 scope_participant: str | None = None
257 scope_trial: str | None = None
258 scope_text: str | None = None
259 # Which screens of a multipart trial go in, by `screen_id` — one id across
260 # trials (OneStop's `Paragraph`, MultiplEYE's `page_1`). `None` is every
261 # screen; a trial showing none of the chosen screens is left out.
262 screens: tuple[str, ...] | None = None
264 def any_table(self) -> bool:
265 return (
266 self.include_fixations
267 or self.include_raw_gaze
268 or self.include_measures
269 or self.include_analysis_family
270 )
272 def table_formats(self) -> list[str]:
273 if self.table_format == "both":
274 return ["csv", "parquet"]
275 return [self.table_format]
277 def figure_formats(self) -> list[str]:
278 formats: list[str] = []
279 if self.include_png:
280 formats.append("png")
281 if self.include_svg:
282 formats.append("svg")
283 if self.include_pdf:
284 formats.append("pdf")
285 if self.include_html:
286 formats.append("html")
287 return formats
289 def raster_formats(self) -> list[str]:
290 """Figure formats that need Kaleido/Chrome (everything but HTML)."""
291 return [f for f in self.figure_formats() if f != "html"]
293 def layer_formats(self) -> list[str]:
294 """Formats for the per-layer breakdown (VIZ-5) — the selected non-HTML
295 figure formats, or SVG when none was picked (vectors suit Illustrator).
296 Empty when separable layers are off."""
297 if not self.separable_layers:
298 return []
299 return self.raster_formats() or ["svg"]
301 def needs_figure(self) -> bool:
302 """Whether the export builds each trial's figure at all (combined figure
303 formats, or the per-layer breakdown)."""
304 return bool(self.figure_formats()) or self.separable_layers
306 def needs_kaleido(self) -> bool:
307 """Whether any figure render goes through Kaleido/Chrome (combined raster
308 formats, or per-layer non-HTML formats)."""
309 return bool(self.raster_formats()) or bool(self.layer_formats())
312@dataclass
313class ExportProgress:
314 total_trials: int
315 finished_trials: int = 0
316 bytes_written: int = 0
317 errors: list[str] = field(default_factory=list)
318 # EXP-24: what the build produced, for the summary under it — every file in
319 # the zip, and the figure files (combined figures and per-layer files)
320 # written and failed, one per format per trial.
321 files_written: int = 0
322 figures_written: int = 0
323 figures_failed: int = 0
324 trials_skipped: int = 0
327@dataclass(frozen=True)
328class ExportSummary:
329 """EXP-24: how a finished bundle reads — ``level`` is ``"success"``,
330 ``"warning"`` (some of it failed) or ``"error"`` (figures were asked for and
331 none was made), and ``expand_errors`` opens the error list for the last."""
333 level: str
334 message: str
335 expand_errors: bool
338#: Session key of the current figure's *Self-contained HTML* choice — the figure's
339#: and the replay's HTML follow it; the bundle asks its own
340#: (``<key_prefix>_html_self_contained``).
341HTML_SELF_CONTAINED_KEY = "export_html_self_contained"
344def html_plotlyjs(self_contained: bool) -> bool | str:
345 """``to_html``'s ``include_plotlyjs`` for a downloaded HTML file: the
346 library embedded (opens offline), or loaded from cdn.plot.ly when opened."""
347 return True if self_contained else "cdn"
350def missing_browser_note(static_formats: bool) -> str:
351 """EXP-24: the bundle's browser prerequisite, when it is unmet, else ``""``.
353 The bundle draws PNG, SVG and PDF through Kaleido on the server, which
354 needs Chrome, Chromium or Edge there; the current figure's PNG and SVG are
355 saved by the reader's own browser and need none — a distinction the panel
356 used to keep to itself, so a bundle on a machine without one built a zip
357 of failures. Asked only when ``static_formats`` are picked.
358 """
359 if not static_formats:
360 return ""
361 from .animation_export import chrome_available
363 if chrome_available():
364 return ""
365 return (
366 "PNG, SVG and PDF bundle figures need Chrome, Chromium or Edge on the "
367 "computer running Scanpath Studio, and none was found, so they will fail. "
368 "Pick **HTML**, which needs no browser — or install one. The current "
369 "figure's PNG and SVG downloads are made by your own browser."
370 )
373def summarize_export(progress: ExportProgress, size_bytes: int) -> ExportSummary:
374 """What a built bundle holds and what failed, in one line (EXP-24).
376 A generic "Ready" over a zip whose figures all failed read as a success,
377 with the failures one click away in an expander. The line counts what was
378 made and what wasn't; a build whose requested figures all failed is an
379 error, with its list open. The zip stays downloadable either way — the
380 files it does hold are good.
381 """
383 def plural(n: int, word: str) -> str:
384 return f"{n:,} {word}{'' if n == 1 else 's'}"
386 parts = [
387 f"{plural(progress.files_written, 'file')} · {size_bytes / 1_048_576:.1f} MB"
388 ]
389 asked = progress.figures_written + progress.figures_failed
390 if asked:
391 parts.append(f"{progress.figures_written:,} of {plural(asked, 'figure')} made")
392 if progress.figures_failed:
393 parts.append(f"{progress.figures_failed:,} failed")
394 if progress.trials_skipped:
395 parts.append(f"{plural(progress.trials_skipped, 'trial')} skipped (no data)")
396 message = " · ".join(parts)
397 if asked and not progress.figures_written:
398 return ExportSummary(
399 "error",
400 f"No figures were made · {message}. The errors are listed below.",
401 True,
402 )
403 if progress.errors:
404 return ExportSummary("warning", f"Partly built · {message}", False)
405 return ExportSummary("success", f"Ready · {message}", False)
408def _safe_id(text: str) -> str:
409 return "".join(c if c.isalnum() or c in "-_." else "_" for c in str(text))
412# Placeholders every pattern gets on top of the trial's own columns.
413_PATTERN_EXTRA_FIELDS = (
414 "artifact",
415 "ext",
416 "n_fixations",
417 "n_words",
418 "reading_time_s",
419 "settings",
420)
421_PLACEHOLDER_RE = re.compile(r"\{([^{}]*)\}")
424def _settings_summary(settings: dict) -> str:
425 """A one-line description of the settings that produced the figure (EXP-2)."""
426 layers = [
427 name
428 for name, key in (
429 ("boxes", "show_words"),
430 ("text", "show_word_labels"),
431 ("fixations", "show_fixations"),
432 ("saccades", "show_saccades"),
433 ("heatmap", "show_heatmap"),
434 )
435 if settings.get(key)
436 ]
437 parts = [f"layers: {', '.join(layers) or 'none'}"]
438 color_by = settings.get("color_by")
439 if color_by and color_by != UNIFORM_COLOR_FIELD:
440 parts.append(f"color by {color_by}")
441 palette = settings.get("palette", DEFAULT_PALETTE)
442 if palette and palette != DEFAULT_PALETTE:
443 parts.append(f"{palette} palette")
444 return " · ".join(parts)
447#: VIZ-36 — the fields that can hold *two* values at once, because an overlay
448#: draws two readings into one frame. Each gains an ``_a`` / ``_b`` variant.
449PAIRED_PATTERN_FIELDS = ("dataset_name", "participant_id", "trial_id", "text_id")
451#: EXP-22 — the tables a pattern can name a field of, as ``{table.field}``, and
452#: the heading each gets in the *Available fields* list. The metadata tables
453#: plus the two data tables' saved fields; a qualified name is what tells two
454#: tables' ``font_size`` apart, and what keeps them out of the plain list.
455TABLE_PATTERN_LABELS = {
456 "participants": "Participants table",
457 "trials": "Trials table",
458 "texts": "Texts table",
459 "fixations": "Fixations table",
460 "words": "Words table",
461}
463#: Columns a data table always carries or the app derives — the trial's own
464#: identity, geometry and timing, which the plain fields already cover or which
465#: are not one value per trial. What is left is what the user kept.
466_CORE_TABLE_COLUMNS = frozenset(
467 {
468 "participant_id",
469 "trial_id",
470 "text_id",
471 "paragraph_id",
472 "unique_trial_id",
473 "unique_text_id",
474 "unique_paragraph_id",
475 "word_id",
476 "text",
477 "line_idx",
478 "x",
479 "y",
480 "width",
481 "height",
482 "screen_id",
483 "screen_index",
484 "canvas_width",
485 "canvas_height",
486 "screen_timestamp_ms",
487 "screen_fixation_id",
488 "duration_ms",
489 "timestamp_ms",
490 "fixation_id",
491 "order_in_trial",
492 "pass_index",
493 "saccade_type",
494 "saccade_amplitude",
495 "eye",
496 "source_file",
497 "TRIAL_INDEX",
498 "trial_index",
499 }
500)
503def _saved_table_fields(frame: pd.DataFrame | None) -> dict:
504 """A data table's saved fields that hold one value for the whole trial.
506 A field that varies within the trial (a word's surprisal, a fixation's
507 pupil size) has no single value to put in a title, so it is not offered."""
508 out: dict = {}
509 if frame is None or getattr(frame, "empty", True):
510 return out
511 for column in frame.columns:
512 name = str(column)
513 if name.startswith("_") or name in _CORE_TABLE_COLUMNS:
514 continue
515 values = frame[column].dropna()
516 if values.empty:
517 continue
518 try:
519 distinct = values.unique()
520 except TypeError: # unhashable cells (lists, dicts)
521 continue
522 if len(distinct) != 1 or isinstance(distinct[0], (list, dict, set, tuple)):
523 continue
524 out[name] = distinct[0]
525 return out
528def table_pattern_fields(
529 trial_words: pd.DataFrame | None,
530 trial_fixations: pd.DataFrame | None,
531 metadata_rows: dict | None = None,
532) -> dict[str, dict]:
533 """``{table: {"table.field": value}}`` for one trial (EXP-22).
535 ``metadata_rows`` is this trial's row of each attached metadata table
536 (``metadata.pattern_rows``); the fixations and AOI tables contribute their
537 saved fields that are constant within the trial. Grouped by table so the
538 *Available fields* list can head each group; :func:`pattern_fields`
539 flattens it."""
540 tables: dict[str, dict] = {}
541 for table, row in (metadata_rows or {}).items():
542 if row:
543 tables[table] = {f"{table}.{name}": value for name, value in row.items()}
544 for table, frame in (("fixations", trial_fixations), ("words", trial_words)):
545 saved = _saved_table_fields(frame)
546 if saved:
547 tables[table] = {f"{table}.{name}": value for name, value in saved.items()}
548 return tables
551def pattern_fields(
552 participant: str,
553 trial: str,
554 trial_words: pd.DataFrame,
555 trial_fixations: pd.DataFrame,
556 settings: dict,
557 combo_row: dict | None = None,
558 dataset_name: str = "",
559 compare_row: dict | None = None,
560 metadata_rows: dict | None = None,
561 column_names: ColumnNames | None = None,
562) -> dict:
563 """Every value a filename / title / caption pattern can substitute.
565 DATA-66: with ``column_names`` (the dataset's map), every field its file
566 named is also reachable under that name — ``{RECORDING_SESSION_LABEL}`` and
567 ``{participant_id}`` both work, so a pattern can be written in either
568 vocabulary.
570 The trial's own combo columns (participant, trial, text, conditions …) plus
571 counts and the settings summary. Values are left raw here; path rendering
572 sanitizes them, while titles and captions want them readable.
574 **VIZ-36 — ``dataset_name`` arrives as an argument, never read from here.**
575 The app knows it as ``data_source_choice`` (since DATA-9 that key *is* the
576 picker's label, and DATA-23's rename re-keys it), but this function is pure
577 and also runs headless under ``api.save_figure_layers`` and ``cli render``,
578 where there is no session at all. Each of the five callers supplies it.
580 ``compare_row`` is the *other* reading in an overlay — two scanpaths in one
581 frame, so a single ``{dataset_name}`` is ambiguous exactly where a title
582 most wants to name both. Every field in :data:`PAIRED_PATTERN_FIELDS` gains
583 an ``_a`` / ``_b`` variant, which are defined **always** (``_b`` empty when
584 there is no second reading) so that a pattern written in compare mode still
585 validates and renders on a single-trial figure instead of erroring on a
586 surface the author cannot see.
588 EXP-22: every attached metadata table's fields and each data table's saved
589 fields join as ``{table.field}`` (:func:`table_pattern_fields`) — qualified,
590 so a trial table's ``font_size`` and a recorded ``font_size`` are both
591 reachable, and none of the plain names above changes.
592 """
593 fields: dict = dict(combo_row or {})
594 for table in table_pattern_fields(
595 trial_words, trial_fixations, metadata_rows
596 ).values():
597 fields.update(table)
598 fields.update(
599 participant_id=participant,
600 trial_id=trial,
601 dataset_name=dataset_name,
602 n_fixations=len(trial_fixations),
603 n_words=len(trial_words),
604 reading_time_s=round(
605 float(
606 pd.to_numeric(trial_fixations.get("duration_ms"), errors="coerce").sum()
607 )
608 / 1000.0,
609 1,
610 )
611 if "duration_ms" in getattr(trial_fixations, "columns", [])
612 else 0.0,
613 settings=_settings_summary(settings),
614 )
615 fields.setdefault("text_id", trial)
616 for name in PAIRED_PATTERN_FIELDS:
617 fields[f"{name}_a"] = fields.get(name, "")
618 fields[f"{name}_b"] = (compare_row or {}).get(name, "")
619 if column_names is not None:
620 # The counts and the settings summary are the app's, whatever a column
621 # of the same canonical name was called in the file.
622 own = {"n_fixations", "n_words", "reading_time_s", "settings", "dataset_name"}
623 named = [field for field in fields if field not in own]
624 for canonical, header in column_names.export_headers(named).items():
625 fields.setdefault(header, fields[canonical])
626 return fields
629def pattern_error(pattern: str, fields: dict) -> str | None:
630 """A human message naming any unknown placeholder, or ``None`` if valid.
632 Validated up front (and shown live in the UI) rather than at export time —
633 discovering a typo after a 200-trial render is the worst place to find it.
634 """
635 known = set(fields) | set(_PATTERN_EXTRA_FIELDS)
636 unknown = [name for name in _PLACEHOLDER_RE.findall(pattern) if name not in known]
637 if not unknown:
638 return None
639 return (
640 f"Unknown field{'s' if len(unknown) > 1 else ''}: "
641 f"{', '.join('{' + u + '}' for u in unknown)}. "
642 + (
643 f"Available: {', '.join('{' + k + '}' for k in sorted(known))}."
644 if len(known) <= 25
645 else f"{len(known)} fields are available; the app's Fields list shows them."
646 )
647 )
650_DRIVE_PREFIX = re.compile(r"^[A-Za-z]:")
653def path_structure_error(pattern: str) -> str | None:
654 """What is wrong with ``pattern``'s own text as a path inside the ZIP, or
655 ``None`` (round 10).
657 The values put into ``{…}`` are sanitized one by one (:func:`render_pattern`),
658 but the text around them is the pattern's: ``../{artifact}.{ext}`` wrote a
659 member outside the archive's root, and an empty pattern one with no name.
660 Every member must be a relative path of named folders ending in a file
661 name, so this refuses an empty pattern, a leading ``/`` or drive, a
662 backslash (a separator to some unzip tools), and an empty, ``.`` or ``..``
663 folder or file name. Checked before any figure is rendered, in the app and
664 by :func:`bulk_export`.
665 """
666 text = str(pattern or "")
667 probe = _PLACEHOLDER_RE.sub("x", text)
668 if not probe.strip():
669 return "The file path pattern is empty."
670 if "\\" in probe:
671 return (
672 "Use `/` between folders: a backslash is a separator to some unzip tools."
673 )
674 if probe.startswith("/") or _DRIVE_PREFIX.match(probe):
675 return "The file path must be relative to the ZIP: start it with a folder or file name."
676 for part in probe.split("/"):
677 if not part.strip():
678 return "The file path has an empty folder or file name (`//`, or a trailing `/`)."
679 if set(part) <= {"."}:
680 return f"`{part}` can't be a folder or file name in the ZIP."
681 return None
684def _path_component(text: str) -> str:
685 """One path segment, sanitized. ``.`` / ``..`` collapse so nothing escapes."""
686 safe = _safe_id(text)
687 return "_" if set(safe) <= {"."} else safe
690def render_pattern(
691 pattern: str,
692 fields: dict,
693 *,
694 as_path: bool = False,
695 multi_segment_fields: tuple = (),
696) -> str:
697 """Substitute ``fields`` into ``pattern``.
699 ``as_path`` sanitizes each substituted value, so a *data* value containing
700 ``/`` or ``..`` becomes one flat segment and can't escape the folder the
701 pattern describes; the pattern's own ``/`` stay real separators.
702 ``multi_segment_fields`` names the tool-controlled fields allowed to expand
703 into several segments (``artifact``, which carries ``layers/<name>``) — each
704 of their segments is still sanitized individually. A missing or null value
705 becomes ``na`` rather than failing the whole export.
706 """
708 def _sub(match: re.Match) -> str:
709 name = match.group(1)
710 value = fields.get(name, "")
711 if value is None or (isinstance(value, float) and pd.isna(value)):
712 value = "na"
713 text = str(value)
714 if not as_path:
715 return text
716 if name in multi_segment_fields:
717 return "/".join(_path_component(part) for part in text.split("/"))
718 return _path_component(text)
720 return _PLACEHOLDER_RE.sub(_sub, pattern)
723def resolve_export_path(
724 pattern: str, fields: dict, *, artifact: str, ext: str, used: set
725) -> str:
726 """The zip path for one artifact, de-duplicated against ``used``.
728 Two trials can render to the same path (a pattern that omits the trial id,
729 say). Writing both would put two entries at one name in the zip and silently
730 lose one, so the second gets a ``-2`` suffix instead. ``used`` is mutated.
731 """
732 path = render_pattern(
733 pattern,
734 {**fields, "artifact": artifact, "ext": ext},
735 as_path=True,
736 multi_segment_fields=("artifact",),
737 ).lstrip("/")
738 if path not in used:
739 used.add(path)
740 return path
741 stem, dot, suffix = path.rpartition(".")
742 base, tail = (stem, f"{dot}{suffix}") if dot else (path, "")
743 n = 2
744 while f"{base}-{n}{tail}" in used:
745 n += 1
746 path = f"{base}-{n}{tail}"
747 used.add(path)
748 return path
751# --- EXP-2 · titles and captions on the exported figure -----------------------
752# Sized bands rather than Plotly's automatic title spacing: the scanpath figure
753# is equal-aspect (`scaleanchor`), so anything that eats into the plot area
754# shrinks the WHOLE plot — and the true-to-scale word labels, computed for the
755# un-shrunk size, then no longer match their boxes. Same constraint the animation
756# transport controls hit; same fix: grow the figure by exactly what the band
757# takes, so the plot region is untouched.
758_TITLE_BAND_PX = 46
759_CAPTION_LINE_PX = 22
760_CAPTION_PAD_PX = 12
763def annotate_figure(fig, *, title: str = "", caption: str = "") -> None:
764 """Stamp ``title`` / ``caption`` onto ``fig`` in place, without shrinking it.
766 The figure grows by the height of each band and its margin grows to match, so
767 the plotting area — and therefore the true-to-scale text — is byte-identical
768 to the untitled figure. Both are drawn as written: ``<b>`` in a title shows
769 as ``<b>``, and only a real newline starts a new caption line.
770 """
771 if not title and not caption:
772 return
773 margin = fig.layout.margin
774 height = fig.layout.height
775 if title:
776 fig.layout.margin.t = (margin.t or 0) + _TITLE_BAND_PX
777 if height:
778 height += _TITLE_BAND_PX
779 fig.layout.height = height
780 fig.update_layout(
781 title=dict(
782 text=_plotly_literal(title),
783 x=0.5,
784 xanchor="center",
785 y=1.0,
786 yanchor="top",
787 pad=dict(t=14),
788 font=dict(size=20),
789 )
790 )
791 if caption:
792 original_bottom = fig.layout.margin.b or 0
793 band = _CAPTION_LINE_PX * (caption.count("\n") + 1) + _CAPTION_PAD_PX
794 fig.layout.margin.b = original_bottom + band
795 if height:
796 fig.layout.height = height + band
797 # Anchored to the plot's bottom edge and pushed into the space just
798 # added, so it never overlaps whatever already lived in that margin.
799 fig.add_annotation(
800 text=_plotly_literal(caption).replace("\n", "<br>"),
801 xref="paper",
802 yref="paper",
803 x=0,
804 y=0,
805 xanchor="left",
806 yanchor="top",
807 yshift=-(original_bottom + _CAPTION_PAD_PX // 2),
808 showarrow=False,
809 align="left",
810 font=dict(size=13, color="#555555"),
811 )
814# DATA-16 (security audit S4). Columns that hold a filesystem path from the
815# machine the app ran on. `image_path` is a `passthrough` meta field on both
816# schemas, so it survives normalization and rides into the exported fixation
817# tables — and a fixations CSV is exactly the file that gets attached to a paper,
818# posted to OSF, or mailed to a collaborator. `/Users/<name>/` discloses the OS
819# account; the rest discloses the directory layout, including where a MultiplEYE
820# corpus lives. The basename still identifies the stimulus, which is all the
821# column is used for downstream.
822#
823# `source_file` is deliberately NOT here: it is an identity label, not a path.
824# `data.source_labels` stores the file's stem, qualified only by the trailing
825# folders that tell two same-named files apart — never the folders they share,
826# so an absolute path's `/Users/<name>/…` prefix does not reach it.
827_PATH_COLUMNS = ("image_path",)
830def strip_local_paths(df: pd.DataFrame) -> pd.DataFrame:
831 """Reduce path-bearing columns to their basename (S4).
833 Returns ``df`` unchanged (the same object) when it carries none of them, so
834 the common case costs one membership test and no copy.
835 """
836 present = [c for c in _PATH_COLUMNS if c in df.columns]
837 if not present:
838 return df
839 out = df.copy()
840 for column in present:
841 values = out[column]
842 # `na_action="ignore"` is load-bearing: pandas evaluates the `other`
843 # argument of `.where` eagerly, over every row including the missing
844 # ones, and since pandas 3 `astype(str)` leaves NaN as a float instead
845 # of stringifying it to "nan" — so the lambda would see a float.
846 out[column] = values.where(
847 values.isna(),
848 values.astype(str).map(
849 lambda text: PurePosixPath(text.replace("\\", "/")).name,
850 na_action="ignore",
851 ),
852 )
853 return out
856#: DATA-66: the artifacts that *are* one of the dataset's tables (rows of it,
857#: perhaps with computed columns added), and which one — so a fixation table's
858#: `x` is named as the fixation file named it and a word table's `x` as the AOI
859#: file did. Every other artifact is derived (a saccade table, a summary, a
860#: character grid) and reuses canonical names for values of its own, so only its
861#: id columns take the file's names (`ColumnNames.identity`).
862_ARTIFACT_TABLE = {
863 "fixations": "fixations",
864 "raw_gaze": "raw_gaze",
865 "measures": "words",
866 "word_measures": "words",
867}
870def _write_table(
871 zf: zipfile.ZipFile,
872 path: str,
873 df: pd.DataFrame,
874 fmt: str,
875 names: ColumnNames | None = None,
876 hidden: set[str] | None = None,
877) -> int:
878 # DATA-49: the pipeline's bookkeeping columns stay out of what is shared.
879 df = strip_local_paths(shareable_frame(df))
880 # DATA-66: and the columns the user's file named go out under those names.
881 df = as_written(df, names, hidden)
882 if fmt == "parquet":
883 buf = io.BytesIO()
884 df.to_parquet(buf, index=False)
885 data = buf.getvalue()
886 else:
887 data = df.to_csv(index=False).encode("utf-8")
888 zf.writestr(path, data)
889 return len(data)
892@contextmanager
893def _figure_renderer(enabled: bool):
894 """Yield ``render(fig, fmt, width, height, scale) -> bytes``.
896 When ``enabled`` and Kaleido starts, every trial's figure is rasterized
897 through one persistent Kaleido browser (``calc_fig_sync``) instead of
898 cold-starting a fresh Chrome on each ``fig.to_image`` call — the cold start
899 is the "Resorting to unclean kill browser." log noise and ~seconds-per-trial
900 latency. Falls back to per-call ``to_image`` if the warm server can't start
901 (or no figures were requested), so behavior is unchanged when Kaleido/Chrome
902 is unavailable — the per-trial failure is still surfaced as an export error.
904 ``enabled`` also holds ``animation_export.KALEIDO_LOCK`` until the server has
905 stopped, since that server is one per process (UX-150).
906 """
907 from .animation_export import KALEIDO_LOCK
909 with KALEIDO_LOCK if enabled else nullcontext():
910 server = None
911 if enabled:
912 try:
913 import kaleido
915 from .animation_export import chromium_browser_path
917 browser_path = chromium_browser_path()
918 if browser_path is not None:
919 kaleido.start_sync_server(path=browser_path, silence_warnings=True)
920 server = kaleido
921 except Exception:
922 server = None
924 def render(fig, fmt: str, width: int, height: int, scale: int) -> bytes:
925 if server is not None:
926 data = server.calc_fig_sync(
927 fig,
928 opts={
929 "format": fmt,
930 "width": int(width),
931 "height": int(height),
932 "scale": scale,
933 },
934 )
935 return bytes(data)
936 return fig.to_image(
937 format=fmt, width=int(width), height=int(height), scale=scale
938 )
940 try:
941 yield render
942 finally:
943 if server is not None:
944 try:
945 server.stop_sync_server(silence_warnings=True)
946 except Exception: # pragma: no cover - best-effort teardown
947 pass
950def render_static_figure_bytes(
951 fig,
952 *,
953 fmt: str,
954 width: int,
955 height: int,
956 scale: float,
957 status_callback: StatusCallback | None = None,
958) -> bytes:
959 """Render one static figure with observable indeterminate job stages."""
960 from .animation_export import CHROME_INSTALL_HINT, chrome_available
962 started = perf_counter()
963 emit_status(
964 status_callback,
965 ExportStage.PREPARING,
966 "Preparing figure and checking export settings…",
967 started_at=started,
968 )
969 try:
970 if not chrome_available():
971 raise RuntimeError(CHROME_INSTALL_HINT)
972 emit_status(
973 status_callback,
974 ExportStage.STARTING_RENDERER,
975 "Starting the Chrome/Kaleido renderer (cold starts can take a few seconds)…",
976 started_at=started,
977 )
978 with _figure_renderer(True) as render:
979 emit_status(
980 status_callback,
981 ExportStage.RASTERIZING,
982 f"Rendering {fmt.upper()}…",
983 started_at=started,
984 )
985 data = render(fig, fmt.lower(), int(width), int(height), float(scale))
986 emit_status(
987 status_callback,
988 ExportStage.FINALIZING,
989 "Finishing the file…",
990 started_at=started,
991 )
992 result = bytes(data)
993 emit_status(
994 status_callback,
995 ExportStage.READY,
996 "Ready to download.",
997 started_at=started,
998 )
999 return result
1000 except Exception as exc:
1001 emit_status(
1002 status_callback,
1003 ExportStage.ERROR,
1004 "Export failed.",
1005 started_at=started,
1006 error=str(exc),
1007 )
1008 raise
1011def _drift_corrected_for_figure(
1012 fix: pd.DataFrame, words: pd.DataFrame, settings: dict
1013) -> tuple[pd.DataFrame, tuple | None]:
1014 """PRE-3 drift correction for one exported figure (EXP-4 / VIZ-24).
1016 Returns ``(figure_fixations, connector_y)``. When no algorithm is selected
1017 (``align_algorithm`` absent / ``"Off"`` — the default) or there is nothing to
1018 correct, hands back the very same frame object and ``None`` — a true no-op,
1019 mirroring ``tabs._drift_corrected``. Otherwise the returned frame has each
1020 fixation's ``y`` snapped to its assigned text line, and ``connector_y``
1021 carries the *original* y values when ``align_connectors`` is on (the faint
1022 original→corrected connector layer).
1024 Deliberate asymmetry: this feeds the **figure only** — the exported tables
1025 (fixations, measures, combined tables) stay uncorrected, because the correction
1026 is a view on the data, not a rewrite of it."""
1027 algorithm = settings.get("align_algorithm")
1028 if (
1029 not algorithm
1030 or str(algorithm) == "Off"
1031 or fix is None
1032 or fix.empty
1033 or words is None
1034 or words.empty
1035 ):
1036 return fix, None
1037 from .alignment import correct # local: pulls in scipy only when used
1039 corrected, _ = correct(fix, words, method=str(algorithm).lower())
1040 connector_y = None
1041 if settings.get("align_connectors") and "y" in fix.columns:
1042 connector_y = tuple(pd.to_numeric(fix["y"], errors="coerce"))
1043 return corrected, connector_y
1046def _plot_config_dict(
1047 participant: str,
1048 trial: str,
1049 canvas_width: int,
1050 canvas_height: int,
1051 x_field: str,
1052 y_field: str,
1053 settings: dict,
1054 *,
1055 screen_id: str | None = None,
1056 drift_applied: bool = False,
1057) -> dict:
1058 selection = {"participant_id": participant, "trial_id": trial}
1059 if screen_id not in (None, ""):
1060 selection[SCREEN_ID] = str(screen_id)
1061 return {
1062 "selection": selection,
1063 "canvas_px": {"width": int(canvas_width), "height": int(canvas_height)},
1064 "axes": {
1065 "x_field": x_field,
1066 "y_field": y_field,
1067 "coordinate_grid": bool(settings.get("show_coordinate_grid", False)),
1068 "coordinate_grid_auto": settings.get("coordinate_grid_spacing") is None,
1069 "coordinate_grid_spacing": settings.get("coordinate_grid_spacing"),
1070 },
1071 "layers": {
1072 "words": settings.get("show_words"),
1073 "word_labels": settings.get("show_word_labels"),
1074 "fixations": settings.get("show_fixations"),
1075 "order_labels": settings.get("show_order"),
1076 "saccades": settings.get("show_saccades"),
1077 "saccade_arrows": settings.get("show_saccade_arrows", False),
1078 "heatmap": settings.get("show_heatmap"),
1079 "raw_gaze": settings.get("show_raw_gaze"),
1080 },
1081 "coloring": {
1082 "color_by": settings.get("color_by"),
1083 "heatmap_metric": settings.get("heatmap_metric"),
1084 "heatmap_style": settings.get("heatmap_style", "Word boxes"),
1085 "fixation_colorscale": settings.get("fixation_colorscale"),
1086 "heatmap_colorscale": settings.get("heatmap_colorscale"),
1087 # VIZ-18 palette · VIZ-17 flat colour · VIZ-15 shape — part of how the
1088 # figure looked, so the manifest records them for reproduction.
1089 "palette": settings.get("palette", DEFAULT_PALETTE),
1090 "fixation_color": settings.get("fixation_color", DEFAULT_FIXATION_COLOR),
1091 "fixation_symbol": settings.get("fixation_symbol", DEFAULT_FIXATION_SYMBOL),
1092 "saccade_color_mode": settings.get("saccade_color_mode", "Uniform"),
1093 # VIZ-31: which reading classes the exported figures actually drew —
1094 # a regressions-only batch has to say so, or the files look like a
1095 # dataset with almost no saccades in it.
1096 "saccade_classes": list(
1097 settings.get("saccade_classes") or SACCADE_CLASS_ORDER
1098 ),
1099 # EXP-4 / VIZ-24: which PRE-3 drift correction produced the exported
1100 # figure ("Off" = none). The exported tables stay uncorrected — the
1101 # manifest is where that split is recorded. Same keys as the 💾 Save
1102 # & restore config (ENG-23). `color_by_line` records the EFFECTIVE
1103 # value: a corrected figure is force-coloured by line, like the
1104 # on-screen static path.
1105 # PRE-21: omitted entirely while drift correction is gated off —
1106 # recording `"drift_correction": "Off"` in every bundle would
1107 # advertise a control the build doesn't have.
1108 **(
1109 {
1110 "drift_correction": str(
1111 settings.get("align_algorithm", "Off") or "Off"
1112 ),
1113 "drift_connectors": bool(settings.get("align_connectors", False)),
1114 }
1115 if drift_correction_enabled()
1116 else {}
1117 ),
1118 "color_by_line": bool(settings.get("color_by_line", False))
1119 or drift_applied,
1120 },
1121 "sizing": {
1122 "marker_size_range": list(settings.get("marker_size_range", [])),
1123 "marker_size_scale": settings.get("marker_size_scale"),
1124 "marker_duration_range": list(settings.get("marker_duration_range") or []),
1125 "duration_size_legend": bool(settings.get("duration_size_legend", True)),
1126 "order_font_size": settings.get("order_font_size"),
1127 },
1128 # Where each legend was placed (Figure & canvas → Legends); absent = Auto.
1129 # Every legend, Auto included, as the settings file writes them: a
1130 # partial section would leave a restoring session's moved legends.
1131 "legends": normalize_legend_layout(settings.get("legend_layout")),
1132 # True-to-scale reading text: records how the word labels were sized so
1133 # the figure can be reproduced exactly (see plots._word_label_font_px).
1134 "text": {
1135 "scale_text_to_boxes": settings.get("scale_text_to_boxes", True),
1136 "line_spacing": settings.get("line_spacing", DEFAULT_LINE_SPACING),
1137 },
1138 # DATA-22 §7 surface 4: the recording setup + how each group is known, so
1139 # an exported figure set records that (say) its monitor size was assumed.
1140 # Sits beside `coloring.drift_correction`, which makes the same kind of
1141 # "what produced these files" statement. Omitted when the source declared
1142 # no setup — an absent key means unknown, which is the truth.
1143 **({"experimental_setup": setup} if (setup := _setup_section()) else {}),
1144 }
1147def _setup_section() -> dict | None:
1148 """The active source's `SetupSnapshot` as a dict, or ``None``.
1150 Imported lazily and defensively: `bulk_export` also runs headlessly (the CLI
1151 and `api.py`), where there is no session to read a snapshot from, and an
1152 export must never fail because it could not describe its own geometry.
1153 """
1154 try:
1155 from scanpath_studio.app import active_setup_snapshot
1157 snapshot = active_setup_snapshot()
1158 except Exception: # pragma: no cover - headless / no session
1159 return None
1160 return snapshot.to_dict() if snapshot is not None else None
1163def _render_scope_picker(
1164 st,
1165 combos: pd.DataFrame,
1166 key_prefix: str,
1167 combos_all: pd.DataFrame | None = None,
1168 selected_participant: str | None = None,
1169 selected_trial: str | None = None,
1170) -> tuple[str, str | None, str | None, str | None, bool]:
1171 """Render the scope radio + dependent selectors.
1173 Returns ``(scope, pid, trial, text, export_unfiltered)``. The whole-dataset
1174 choice lives inside the "Trials to include" radio (an extra "All" option that
1175 ignores the trial filters) rather than as a separate checkbox.
1176 """
1177 # Build the ordered radio: label -> (scope, export_unfiltered). Both "All"
1178 # (the whole dataset, ignoring the trial filters) and "All filtered trials"
1179 # (the current filter selection) are always offered — they coincide only
1180 # when no filter is active.
1181 options_map: dict[str, tuple[str, bool]] = {
1182 "This trial": ("trial", False),
1183 "All": ("all", True),
1184 "All filtered trials": ("all", False),
1185 }
1186 options_map["All trials of one participant"] = ("participant", False)
1187 options_map["All trials of one text"] = ("text", False)
1189 # Default to the filtered subset (respect what the user narrowed to).
1190 default_index = 2
1191 scope_label = st.radio(
1192 "Trials to include",
1193 options=list(options_map),
1194 index=default_index,
1195 key=f"{key_prefix}_scope",
1196 # The stored value stays "All"; only what the radio shows says more.
1197 format_func=lambda label: "All, ignoring filters" if label == "All" else label,
1198 horizontal=True,
1199 help="Choose a subset. All ignores active filters.",
1200 label_visibility="collapsed",
1201 )
1202 scope, export_unfiltered = options_map[scope_label]
1203 active = combos_all if (export_unfiltered and combos_all is not None) else combos
1205 scope_participant: str | None = None
1206 scope_trial: str | None = None
1207 scope_text: str | None = None
1208 text_col = (
1209 "unique_text_id"
1210 if "unique_text_id" in active.columns
1211 else ("text_id" if "text_id" in active.columns else None)
1212 )
1214 if scope == "trial" and not active.empty:
1215 if selected_participant is not None and selected_trial is not None:
1216 scope_participant = str(selected_participant)
1217 scope_trial = str(selected_trial)
1218 else:
1219 participants = sorted(
1220 active["participant_id"].dropna().astype(str).unique()
1221 )
1222 scope_participant = panel_field(
1223 st,
1224 "selectbox",
1225 "Participant",
1226 options=participants,
1227 key=f"{key_prefix}_scope_pid",
1228 )
1229 trials_for_pid = (
1230 active.loc[
1231 active["participant_id"].astype(str) == str(scope_participant),
1232 "trial_id",
1233 ]
1234 .astype(str)
1235 .unique()
1236 )
1237 scope_trial = panel_field(
1238 st,
1239 "selectbox",
1240 "Trial",
1241 options=sorted(trials_for_pid),
1242 key=f"{key_prefix}_scope_trial",
1243 )
1244 elif scope == "participant" and not active.empty:
1245 participants = sorted(active["participant_id"].dropna().astype(str).unique())
1246 scope_participant = panel_field(
1247 st,
1248 "selectbox",
1249 "Participant",
1250 options=participants,
1251 key=f"{key_prefix}_scope_pid",
1252 )
1253 elif scope == "text" and not active.empty:
1254 if text_col is None:
1255 st.info("This dataset has no text ids, so it can't be exported by text.")
1256 else:
1257 texts = sorted(active[text_col].dropna().astype(str).unique())
1258 scope_text = panel_field(
1259 st,
1260 "selectbox",
1261 "Text",
1262 options=texts,
1263 key=f"{key_prefix}_scope_text",
1264 )
1266 # Close the Scope section with a live count of what will be exported.
1267 n_export = len(
1268 _scope_frame(active, scope, scope_participant, scope_trial, scope_text)
1269 )
1270 n_total = len(combos_all) if combos_all is not None else len(combos)
1271 st.caption(f"**{n_export:,}** of **{n_total:,}** trials will be exported.")
1273 return scope, scope_participant, scope_trial, scope_text, export_unfiltered
1276def _preview_fields(combos: pd.DataFrame) -> dict:
1277 """Stand-in field values for the live pattern preview (EXP-1/EXP-2).
1279 Uses the first trial in scope, so the preview is a path the user will
1280 actually get rather than a made-up example.
1281 """
1282 row = combos.iloc[0].to_dict() if combos is not None and not combos.empty else {}
1283 fields = dict(row)
1284 fields.setdefault("participant_id", "p01")
1285 fields.setdefault("trial_id", "t01")
1286 fields.setdefault("text_id", fields["trial_id"])
1287 fields.update(
1288 n_fixations=123,
1289 n_words=45,
1290 reading_time_s=18.4,
1291 settings="layers: fixations, saccades, text",
1292 )
1293 return fields
1296def _prune_stale_fields(state, state_key: str, names: list[str]) -> None:
1297 """Drop field names the attached table no longer has from a stored picker
1298 selection — the house `controls._drop_stale_multi` pattern.
1300 Two states must stay apart. An **empty** stored list is the user clearing
1301 the picker, which leaves the table out of the bundle, so it is kept as is.
1302 A non-empty list that names **only** stale fields is what a replaced table
1303 leaves behind; Streamlit would filter it to `[]` and read it as that same
1304 omission, so it is removed and the picker starts again on every field.
1305 """
1306 stored = state.get(state_key)
1307 if not isinstance(stored, (list, tuple)) or not stored:
1308 return
1309 kept = [name for name in stored if name in names]
1310 if not kept:
1311 state.pop(state_key, None)
1312 elif list(stored) != kept:
1313 state[state_key] = kept
1316def _render_metadata_field_picker(key_prefix: str):
1317 """DATA-20 milestone 10 — which participant fields ride along in the bundle.
1319 Renders nothing when no table is attached, and returns ``None`` — "no
1320 restriction" — while every field is still selected, so an export made
1321 without touching this control is byte-identical to one made before the
1322 control existed.
1324 It lives with the export options rather than beside the table on the 🗂️ Data
1325 page because it is a decision about *this bundle*: the same attached table
1326 can reasonably ship its full detail to a collaborator and only a group label
1327 to a public repository.
1328 """
1329 import streamlit as st
1331 from scanpath_studio import metadata as md
1333 attached = md.active()
1334 if attached is None or not attached.fields:
1335 return None
1336 names = [field.name for field in attached.fields]
1337 labels = {field.name: field.label for field in attached.fields}
1338 state_key = f"{key_prefix}_meta_fields"
1339 _prune_stale_fields(st.session_state, state_key, names)
1340 chosen = panel_field(
1341 st,
1342 "multiselect",
1343 "Participant fields to include",
1344 options=names,
1345 default=names,
1346 format_func=lambda name: labels.get(name, name),
1347 key=state_key,
1348 persist_state="session",
1349 help="The participant id is always kept.",
1350 )
1351 ordered = tuple(name for name in names if name in set(chosen))
1352 return None if len(ordered) == len(names) else ordered
1355def _render_trial_metadata_field_picker(key_prefix: str):
1356 """DATA-29 — which trial fields ride along, the twin of the picker above.
1358 Same contract in every respect: nothing rendered and ``None`` returned when
1359 no trial table is attached or while every field is still chosen, so an
1360 export made without touching it is unchanged.
1361 """
1362 import streamlit as st
1364 from scanpath_studio import metadata as md
1366 attached = md.active_trials()
1367 if attached is None or not attached.fields:
1368 return None
1369 names = [field.name for field in attached.fields]
1370 labels = {field.name: field.label for field in attached.fields}
1371 state_key = f"{key_prefix}_trial_meta_fields"
1372 _prune_stale_fields(st.session_state, state_key, names)
1373 chosen = panel_field(
1374 st,
1375 "multiselect",
1376 "Trial fields to include",
1377 options=names,
1378 default=names,
1379 format_func=lambda name: labels.get(name, name),
1380 key=state_key,
1381 persist_state="session",
1382 help="The trial id is always kept.",
1383 )
1384 ordered = tuple(name for name in names if name in set(chosen))
1385 return None if len(ordered) == len(names) else ordered
1388def _render_text_metadata_field_picker(key_prefix: str):
1389 """Which text fields ride along — third grain, twin of the two above.
1391 Same contract: nothing rendered and ``None`` returned when no text table
1392 is attached or while every field is still chosen, so an export made
1393 without touching it is unchanged.
1394 """
1395 import streamlit as st
1397 from scanpath_studio import metadata as md
1399 attached = md.active_texts()
1400 if attached is None or not attached.fields:
1401 return None
1402 names = [field.name for field in attached.fields]
1403 labels = {field.name: field.label for field in attached.fields}
1404 state_key = f"{key_prefix}_text_meta_fields"
1405 _prune_stale_fields(st.session_state, state_key, names)
1406 chosen = panel_field(
1407 st,
1408 "multiselect",
1409 "Text fields to include",
1410 options=names,
1411 default=names,
1412 format_func=lambda name: labels.get(name, name),
1413 key=state_key,
1414 persist_state="session",
1415 help="The text id is always kept.",
1416 )
1417 ordered = tuple(name for name in names if name in set(chosen))
1418 return None if len(ordered) == len(names) else ordered
1421def _render_naming_options(st, combos: pd.DataFrame, key_prefix: str):
1422 """The compact **File naming** block: EXP-1's path pattern.
1424 Every pattern is validated and previewed against the first trial in scope as
1425 it's typed — finding a typo after a 200-trial render is the worst possible
1426 place to find it. Returns ``path_pattern``; an invalid pattern falls back to
1427 its default so a bad keystroke can't produce a broken zip.
1429 The title/caption pair used to live here too (EXP-2) but moved to the
1430 Scanpath rail's **📐 Figure & canvas** group (EXP-5), so it's visible on the
1431 live figure and not just at export time; `render_export_options` reads it
1432 back from there instead of keeping a second, possibly-diverging copy.
1433 """
1434 fields = _preview_fields(combos)
1435 available = ", ".join(
1436 f"`{{{k}}}`" for k in sorted(set(fields) | set(_PATTERN_EXTRA_FIELDS))
1437 )
1439 def _pattern_input(label: str, default: str, key: str, help_text: str) -> str:
1440 value = panel_field(
1441 st, "text_input", label, value=default, key=key, help=help_text
1442 )
1443 error = pattern_error(value, fields) or path_structure_error(value)
1444 if error:
1445 st.error(error)
1446 return default
1447 return value
1449 heading, fields_slot = st.columns([5, 1], vertical_alignment="bottom")
1450 heading.markdown("### File naming")
1451 with fields_slot.popover(
1452 "Fields", help="Placeholders available in the path pattern."
1453 ):
1454 st.markdown(available)
1455 path_pattern = _pattern_input(
1456 "File path pattern",
1457 DEFAULT_PATH_PATTERN,
1458 f"{key_prefix}_path_pattern",
1459 "Path inside the ZIP. Use `/` for folders and `{…}` placeholders.",
1460 )
1461 st.caption(
1462 "Example: `"
1463 + resolve_export_path(
1464 path_pattern, fields, artifact="figure", ext="png", used=set()
1465 )
1466 + "`"
1467 )
1468 return path_pattern
1471def render_export_options(
1472 st_module,
1473 combos: pd.DataFrame,
1474 key_prefix: str = "export",
1475 combos_all: pd.DataFrame | None = None,
1476 title_pattern: str = "",
1477 caption_pattern: str = "",
1478 selected_participant: str | None = None,
1479 selected_trial: str | None = None,
1480 screen_options: list[str] | None = None,
1481) -> ExportOptions:
1482 """Render the bulk-export options UI and return a populated ExportOptions.
1484 ``screen_options`` are the dataset's screen ids (:func:`screen_choices`);
1485 when there are any, a *Screens* picker narrows the bundle to some of them.
1487 ``combos`` is the currently filtered trial pool; ``combos_all`` (when given)
1488 is the whole loaded dataset. Picking the "All" scope switches the scope
1489 picker — and the export itself — to ``combos_all`` so the trial filters
1490 are ignored. ``title_pattern``/``caption_pattern`` come from the Scanpath
1491 rail's **📐 Figure & canvas** → *Title* / *Caption* (EXP-5) —
1492 this panel no longer has its own copy of that setting.
1493 """
1494 st = st_module
1495 # No expander — the options are always displayed.
1496 with st.container():
1497 st.markdown("### Trials to include")
1498 # The whole-dataset choice lives inside the scope radio.
1499 (
1500 scope,
1501 scope_pid,
1502 scope_trial,
1503 scope_text,
1504 export_unfiltered,
1505 ) = _render_scope_picker(
1506 st,
1507 combos,
1508 key_prefix,
1509 combos_all=combos_all,
1510 selected_participant=selected_participant,
1511 selected_trial=selected_trial,
1512 )
1513 screens = _render_screen_picker(st, screen_options or [], key_prefix)
1515 # Figures are the headline artifact, so they lead with a single
1516 # multi-select of formats (pills) rather than a column of checkboxes.
1517 st.markdown("### Figure formats")
1518 fig_formats = (
1519 panel_field(
1520 st,
1521 "pills",
1522 "Formats",
1523 options=["PDF", "SVG", "PNG", "HTML"],
1524 selection_mode="multi",
1525 default=["PDF"],
1526 key=f"{key_prefix}_figfmts",
1527 help="PDF/SVG are vector, PNG is raster, and HTML is interactive. "
1528 "The bundle draws PDF, SVG and PNG with Chrome, Chromium or "
1529 "Edge; HTML needs no browser.",
1530 )
1531 or []
1532 )
1533 include_pdf = "PDF" in fig_formats
1534 include_svg = "SVG" in fig_formats
1535 include_png = "PNG" in fig_formats
1536 include_html = "HTML" in fig_formats
1537 # Asked only while HTML is picked, as for the current figure above.
1538 html_self_contained = include_html and bool(
1539 panel_field(
1540 st,
1541 "checkbox",
1542 "Self-contained HTML (opens offline, larger file)",
1543 display="Self-contained HTML",
1544 value=False,
1545 key=f"{key_prefix}_html_self_contained",
1546 persist_state="session",
1547 help="Tick to make the bundle's HTML figures open without an "
1548 "internet connection: each carries the Plotly library, about "
1549 "4.8 MB more. Unticked, they load it from cdn.plot.ly when opened.",
1550 )
1551 )
1552 # EXP-24: the bundle's static figures need a browser on the server,
1553 # which the current figure's PNG/SVG do not — said only when it is
1554 # missing, and only while one of those formats is picked.
1555 if note := missing_browser_note(include_pdf or include_svg or include_png):
1556 st.warning(note, icon=ICONS["warning"])
1557 # Only surface the scale stepper when PNG is on, and keep it narrow.
1558 # `width` does here what the old `st.columns([1, 3])` did: a 1–4 stepper
1559 # stretched across the whole field column reads as a text box. UX-69
1560 # dropped the columns because they nested inside the row's own.
1561 if include_png:
1562 png_scale = panel_field(
1563 st,
1564 "number_input",
1565 "PNG scale",
1566 min_value=1,
1567 max_value=4,
1568 value=2,
1569 width=140,
1570 key=f"{key_prefix}_scale",
1571 help="Higher values increase quality and file size.",
1572 )
1573 else:
1574 png_scale = int(st.session_state.get(f"{key_prefix}_scale", 2))
1576 st.markdown("### Also include")
1577 # VIZ-5: per-layer figure breakdown for publication editing.
1578 separable_layers = panel_field(
1579 st,
1580 "toggle",
1581 "Separable layers",
1582 value=False,
1583 key=f"{key_prefix}_layers",
1584 help="Export each visual layer as a separate file.",
1585 )
1586 # Layers are static vectors/rasters — HTML can't be split. When no
1587 # vector/raster format is picked, they fall back to SVG (which needs
1588 # Kaleido/Chrome, unlike the browser-free HTML the user chose), so warn.
1589 if separable_layers and not (include_png or include_svg or include_pdf):
1590 st.caption(
1591 f"{ICONS['warning']} Separable layers export as **SVG** (a static vector needing "
1592 "Chrome/Kaleido) — HTML figures can't be split. Pick SVG/PDF/PNG "
1593 "above to choose the layer format."
1594 )
1595 include_plot_config = panel_field(
1596 st,
1597 "toggle",
1598 "Settings file (JSON)",
1599 value=True,
1600 key=f"{key_prefix}_cfg",
1601 help="Include the figure's settings file.",
1602 )
1603 include_annotations = panel_field(
1604 st,
1605 "toggle",
1606 "Annotations (JSON)",
1607 value=False,
1608 key=f"{key_prefix}_annotations",
1609 help="Include the exported trials' favorites, tags and notes as one "
1610 f"annotations.json — the file {ICONS['view_data']} Data Management → Annotations imports.",
1611 )
1612 tabular = (
1613 panel_field(
1614 st,
1615 "pills",
1616 "Tabular data",
1617 # The measure family is computed by the app, and held back
1618 # with the other computed measures.
1619 options=[
1620 "Fixations",
1621 "Raw gaze",
1622 "Word measures",
1623 *(["Full measure family"] if computed_measures_enabled() else []),
1624 ],
1625 selection_mode="multi",
1626 default=[],
1627 key=f"{key_prefix}_tabular",
1628 help="Choose the data tables to include. Word measures are the "
1629 "reading measures the dataset brought; none are computed.",
1630 )
1631 or []
1632 )
1633 include_fixations = "Fixations" in tabular
1634 include_raw_gaze = "Raw gaze" in tabular
1635 include_measures = "Word measures" in tabular
1636 include_analysis_family = "Full measure family" in tabular
1637 any_table = bool(tabular)
1638 if any_table:
1639 table_format = (
1640 panel_field(
1641 st,
1642 "segmented_control",
1643 "Table format",
1644 options=["csv", "parquet", "both"],
1645 format_func=lambda value: {
1646 "csv": "CSV",
1647 "parquet": "Parquet",
1648 "both": "Both",
1649 }[value],
1650 default="csv",
1651 key=f"{key_prefix}_fmt",
1652 )
1653 or "csv"
1654 )
1655 combine_trials = panel_field(
1656 st,
1657 "toggle",
1658 "Combine all trials into one file per table",
1659 value=False,
1660 key=f"{key_prefix}_combine",
1661 help="Write each table once, with every exported trial stacked "
1662 "in it, under aggregate/ — instead of one file per trial.",
1663 )
1664 else:
1665 table_format = str(st.session_state.get(f"{key_prefix}_fmt", "csv"))
1666 combine_trials = bool(st.session_state.get(f"{key_prefix}_combine", False))
1668 metadata_fields = _render_metadata_field_picker(key_prefix)
1669 trial_metadata_fields = _render_trial_metadata_field_picker(key_prefix)
1670 text_metadata_fields = _render_text_metadata_field_picker(key_prefix)
1671 path_pattern = _render_naming_options(st, combos, key_prefix)
1672 if title_pattern or caption_pattern:
1673 st.caption(
1674 "Title and caption on the figure — set on the Scanpath rail's "
1675 f"**{ICONS['figure']} Figure & canvas** → *Title* / *Caption*, and "
1676 "applied here too."
1677 )
1679 return ExportOptions(
1680 include_png=include_png,
1681 include_svg=include_svg,
1682 include_pdf=include_pdf,
1683 include_html=include_html,
1684 html_self_contained=html_self_contained,
1685 include_plot_config=include_plot_config,
1686 include_annotations=include_annotations,
1687 include_fixations=include_fixations,
1688 include_raw_gaze=include_raw_gaze,
1689 include_measures=include_measures,
1690 include_analysis_family=include_analysis_family,
1691 combine_trials=combine_trials,
1692 separable_layers=separable_layers,
1693 table_format=table_format,
1694 png_scale=int(png_scale),
1695 path_pattern=path_pattern,
1696 title_pattern=title_pattern,
1697 caption_pattern=caption_pattern,
1698 # VIZ-36: this panel only runs inside the app, so the picker's label is
1699 # available; `pattern_fields` itself stays session-free.
1700 dataset_name=_session_dataset_name(),
1701 metadata_fields=metadata_fields,
1702 trial_metadata_fields=trial_metadata_fields,
1703 text_metadata_fields=text_metadata_fields,
1704 export_unfiltered=export_unfiltered,
1705 scope=scope,
1706 scope_participant=scope_pid,
1707 scope_trial=scope_trial,
1708 scope_text=scope_text,
1709 screens=screens,
1710 )
1713def _render_screen_picker(
1714 st, screen_options: list[str], key_prefix: str
1715) -> tuple[str, ...] | None:
1716 """The *Screens* row: which screens of each multipart trial go in. Nothing
1717 picked is every screen (``None``); drawn only for a dataset with screens."""
1718 if not screen_options:
1719 return None
1720 key = f"{key_prefix}_screens"
1721 # A pick from another dataset names screens this one does not have.
1722 held = st.session_state.get(key)
1723 if held is not None:
1724 kept = [screen for screen in held if screen in screen_options]
1725 if kept != list(held):
1726 st.session_state[key] = kept
1727 picked = (
1728 panel_field(
1729 st,
1730 "multiselect",
1731 "Screens",
1732 options=screen_options,
1733 key=key,
1734 placeholder="All screens",
1735 help="Export only these screens of each trial. Leave empty for "
1736 "every screen; a trial that shows none of them is left out.",
1737 )
1738 or []
1739 )
1740 return tuple(picked) if picked else None
1743def _session_dataset_name() -> str:
1744 """The dataset picker's label, for the options UI only (VIZ-36).
1746 Guarded because `export` is imported headless by `api` and `cli`, where
1747 there is no session; the value only ever reaches `ExportOptions`, never
1748 `pattern_fields`, which takes it as an argument by design.
1749 """
1750 import streamlit as st
1752 try:
1753 return str(st.session_state.get("data_source_choice") or "")
1754 except Exception:
1755 return ""
1758def _scope_frame(
1759 combos: pd.DataFrame,
1760 scope: str,
1761 scope_participant: str | None,
1762 scope_trial: str | None,
1763 scope_text: str | None,
1764) -> pd.DataFrame:
1765 """Filter combos to the chosen scope (pure helper, no ExportOptions needed)."""
1766 if scope == "trial" and scope_participant and scope_trial:
1767 return combos[
1768 (combos["participant_id"].astype(str) == str(scope_participant))
1769 & (combos["trial_id"].astype(str) == str(scope_trial))
1770 ]
1771 if scope == "participant" and scope_participant:
1772 return combos[combos["participant_id"].astype(str) == str(scope_participant)]
1773 if scope == "text" and scope_text:
1774 text_col = (
1775 "unique_text_id"
1776 if "unique_text_id" in combos.columns
1777 else ("text_id" if "text_id" in combos.columns else None)
1778 )
1779 if text_col is None:
1780 return combos
1781 return combos[combos[text_col].astype(str) == str(scope_text)]
1782 return combos
1785def _apply_scope(combos: pd.DataFrame, options: ExportOptions) -> pd.DataFrame:
1786 """Filter combos according to options.scope."""
1787 return _scope_frame(
1788 combos,
1789 options.scope,
1790 options.scope_participant,
1791 options.scope_trial,
1792 options.scope_text,
1793 )
1796@dataclass(frozen=True)
1797class ComparisonSide:
1798 """One half of an exported comparison pair (CMP-8 §6).
1800 ``dataset`` is the corpus label — ``None`` for the active dataset, which is
1801 what every same-dataset comparison passes. ``participant`` / ``trial`` are
1802 the **real** ids, as the corpus spells them: the ``dataset · pid`` namespace
1803 exists only inside the figure's throwaway frames, and an exported table that
1804 used it would not match its own corpus.
1805 """
1807 participant: str
1808 trial: str
1809 words: pd.DataFrame
1810 fixations: pd.DataFrame
1811 dataset: str | None = None
1812 setup: dict | None = None
1814 @property
1815 def slug(self) -> str:
1816 return f"{_safe_id(self.participant)}__{_safe_id(self.trial)}"
1818 def stamped(self, side: str | None = None) -> tuple[pd.DataFrame, pd.DataFrame]:
1819 """Both frames with a ``dataset`` column, so the pair's tables are readable.
1821 Two corpora can hold the same ``(participant_id, trial_id)``; without
1822 this column the rows in ``fixations.csv`` would be indistinguishable.
1823 ``side`` (CMP-22) also stamps a ``scanpath`` column, ``"A"`` or ``"B"``:
1824 B can now be A's own trial, whose rows match A's on every other column.
1825 """
1826 label = self.dataset or "(this dataset)"
1827 out = []
1828 for frame in (self.words, self.fixations):
1829 if frame is None or frame.empty:
1830 out.append(frame)
1831 continue
1832 stamped = frame.copy()
1833 stamped["dataset"] = label
1834 if side is not None:
1835 stamped["scanpath"] = side
1836 out.append(stamped)
1837 return out[0], out[1]
1839 def manifest(self) -> dict:
1840 return {
1841 "source": self.dataset,
1842 "participant": str(self.participant),
1843 "trial": str(self.trial),
1844 "setup": self.setup,
1845 }
1848def pair_export(
1849 fig,
1850 side_a: ComparisonSide,
1851 side_b: ComparisonSide,
1852 *,
1853 canvas_width: int,
1854 canvas_height: int,
1855 x_field: str,
1856 y_field: str,
1857 settings: dict,
1858 options: ExportOptions,
1859 status_callback: StatusCallback | None = None,
1860 column_names: Mapping[str, ColumnNames] | None = None,
1861) -> bytes:
1862 """Zip one comparison **pair** — figure, manifest, and both scanpaths' tables.
1864 ``column_names`` (DATA-66) is the active dataset's map per table. It names
1865 the tables' columns only when both scanpaths come from that dataset: a pair
1866 across two datasets shares no file names, so its tables keep the internal
1867 ones, which both datasets understand.
1869 CMP-8 §6. An exported cross-dataset figure is unreproducible on its own:
1870 nothing in the image records where B came from. The bundle writes the bulk
1871 exporter's per-trial shape for the pair instead of for one trial, and the
1872 ``datasets`` block in ``plot_config.json`` is what makes it reproducible —
1873 it names both sources, both trials, and both recording setups::
1875 <A>__vs__<B>/
1876 ├─ figure.<fmt>
1877 ├─ plot_config.json
1878 ├─ fixations.csv
1879 └─ measures.csv
1881 ``options`` is the ordinary `ExportOptions` — the pair is one more "trial
1882 folder" as far as the writer is concerned, so the formats, table format and
1883 figure toggles all mean what they already mean. Deliberately unchanged from
1884 bulk export: the tables stay **uncorrected** by drift correction (EXP-4 /
1885 VIZ-24), which the manifest records.
1886 """
1887 folder = f"{side_a.slug}__vs__{side_b.slug}"
1888 buffer = io.BytesIO()
1889 started = perf_counter()
1890 emit_status(
1891 status_callback,
1892 ExportStage.PREPARING,
1893 "Preparing the comparison pair…",
1894 started_at=started,
1895 )
1896 with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as zf:
1897 for fmt in options.figure_formats():
1898 if fig is None:
1899 break
1900 width = int(getattr(fig.layout, "width", None) or canvas_width)
1901 height = int(getattr(fig.layout, "height", None) or canvas_height)
1902 if fmt == "html":
1903 data = fig.to_html(
1904 include_plotlyjs=html_plotlyjs(options.html_self_contained),
1905 full_html=True,
1906 config={**PLOTLY_CONFIG},
1907 ).encode("utf-8")
1908 else:
1909 data = render_static_figure_bytes(
1910 fig,
1911 fmt=fmt,
1912 width=width,
1913 height=height,
1914 scale=3 if fmt == "png" else 1,
1915 status_callback=status_callback,
1916 )
1917 zf.writestr(f"{folder}/figure.{fmt}", data)
1919 words_a, fix_a = side_a.stamped("A")
1920 words_b, fix_b = side_b.stamped("B")
1921 maps = (
1922 {
1923 table: names
1924 for table, names in (column_names or {}).items()
1925 if names is not None and names.entries
1926 }
1927 if side_b.dataset in (None, side_a.dataset)
1928 else {}
1929 )
1930 # AN-32 / EXP-23: the measures each side's dataset brought.
1931 measures = [
1932 words.assign(scanpath=side)
1933 for side, words in (("A", words_a), ("B", words_b))
1934 if words is not None and not words.empty and brought_reading_measures(words)
1935 ]
1936 tables = {
1937 "fixations": pd.concat([fix_a, fix_b], ignore_index=True)
1938 if options.include_fixations
1939 else None,
1940 "words": pd.concat(measures, ignore_index=True)
1941 if options.include_measures and measures
1942 else None,
1943 }
1944 files = {"fixations": "fixations", "words": "measures"}
1945 written = {
1946 table: written_columns(shareable_frame(frame), maps[table])
1947 for table, frame in tables.items()
1948 if frame is not None and table in maps
1949 }
1950 for fmt in options.table_formats():
1951 for table, frame in tables.items():
1952 if frame is not None:
1953 _write_table(
1954 zf,
1955 f"{folder}/{files[table]}.{fmt}",
1956 frame,
1957 fmt,
1958 maps.get(table),
1959 )
1960 if any(written.values()):
1961 zf.writestr(
1962 f"{folder}/columns.json",
1963 json.dumps(columns_manifest(written), indent=2),
1964 )
1966 config = _plot_config_dict(
1967 side_a.participant,
1968 side_a.trial,
1969 canvas_width,
1970 canvas_height,
1971 x_field,
1972 y_field,
1973 settings,
1974 )
1975 # The block that makes the pair reproducible — without it the figure
1976 # names two readers and no way to find either of them again.
1977 config["datasets"] = {"a": side_a.manifest(), "b": side_b.manifest()}
1978 zf.writestr(
1979 f"{folder}/plot_config.json", json.dumps(config, indent=2).encode("utf-8")
1980 )
1981 emit_status(
1982 status_callback,
1983 ExportStage.READY,
1984 "Comparison pair ready.",
1985 started_at=started,
1986 )
1987 return buffer.getvalue()
1990def _selected_metadata_columns(frame, fields: tuple[str, ...] | None):
1991 """``frame`` narrowed to ``fields`` (+ the reader id), or ``None`` to drop it.
1993 ``fields is None`` keeps every column — the default, so nothing changes for
1994 a caller that never heard of the opt-out. An **empty** tuple means the user
1995 cleared the picker: the table is left out of the bundle entirely, rather
1996 than shipped as a bare list of reader ids.
1997 """
1998 if frame is None or fields is None:
1999 return frame
2000 if not fields:
2001 return None
2002 keep = ["participant_id"] if "participant_id" in frame.columns else []
2003 keep += [name for name in fields if name in frame.columns and name not in keep]
2004 return frame[keep] if keep else None
2007def _selected_trial_metadata_columns(frame, fields: tuple[str, ...] | None):
2008 """``frame`` narrowed to ``fields`` (+ the trial key), or ``None`` to drop it.
2010 The trial twin of :func:`_selected_metadata_columns`. The key it always
2011 keeps is whichever the table was attached with — ``trial_id`` alone, or
2012 ``participant_id`` beside it — because a trial table shipped without its
2013 key cannot be joined back to anything.
2014 """
2015 if frame is None or fields is None:
2016 return frame
2017 if not fields:
2018 return None
2019 keep = [name for name in ("participant_id", "trial_id") if name in frame.columns]
2020 keep += [name for name in fields if name in frame.columns and name not in keep]
2021 return frame[keep] if keep else None
2024def _selected_text_metadata_columns(frame, fields: tuple[str, ...] | None):
2025 """``frame`` narrowed to ``fields`` (+ the text key), or ``None`` to drop it.
2027 The text twin of :func:`_selected_metadata_columns`/
2028 :func:`_selected_trial_metadata_columns` — a text table shipped without
2029 its key cannot be joined back to anything.
2030 """
2031 if frame is None or fields is None:
2032 return frame
2033 if not fields:
2034 return None
2035 keep = ["text_id"] if "text_id" in frame.columns else []
2036 keep += [name for name in fields if name in frame.columns and name not in keep]
2037 return frame[keep] if keep else None
2040def _rows_in_scope(
2041 frame,
2042 *,
2043 pairs: set[tuple[str, str]],
2044 texts: set[str],
2045 grain: str,
2046):
2047 """``frame``'s rows about what the bundle exports, or ``None`` when none are.
2049 A bundle's metadata describes only its own readers, trials and texts:
2050 ``grain`` ``"participant"`` keeps the readers in ``pairs``, ``"trial"`` the
2051 readings — by the (reader, trial) pair when the table has a reader column,
2052 else by trial id — and ``"text"`` the texts in ``texts``. Ids compare as
2053 text, as the metadata module matches them. A table with no key column, or
2054 no matching row, is left out rather than shipped whole.
2055 """
2056 if frame is None or frame.empty:
2057 return None
2058 if grain == "participant":
2059 if "participant_id" not in frame.columns:
2060 return None
2061 mask = frame["participant_id"].astype(str).isin({p for p, _ in pairs})
2062 elif grain == "trial":
2063 if "trial_id" not in frame.columns:
2064 return None
2065 if "participant_id" in frame.columns:
2066 keys = pd.Series(
2067 list(
2068 zip(
2069 frame["participant_id"].astype(str),
2070 frame["trial_id"].astype(str),
2071 strict=True,
2072 )
2073 ),
2074 index=frame.index,
2075 dtype=object,
2076 )
2077 mask = keys.isin(pairs)
2078 else:
2079 mask = frame["trial_id"].astype(str).isin({t for _, t in pairs})
2080 else:
2081 if "text_id" not in frame.columns:
2082 return None
2083 mask = frame["text_id"].astype(str).isin(texts)
2084 kept = frame[mask]
2085 return None if kept.empty else kept
2088def _unit_text_ids(combo_row: dict, *frames: pd.DataFrame) -> set[str]:
2089 """Every text id one exported reading carries — its combo row's, and the
2090 ``unique_text_id`` / ``text_id`` values in its own rows — as text."""
2091 found: set[str] = set()
2092 value = combo_row.get("text_id")
2093 if value is not None and not pd.isna(value):
2094 found.add(str(value))
2095 for frame in frames:
2096 if frame is None or frame.empty:
2097 continue
2098 for column in ("unique_text_id", "text_id"):
2099 if column in frame.columns:
2100 found.update(frame[column].dropna().astype(str).unique())
2101 return found
2104def _session_text_metadata():
2105 """The attached text table, when running inside the app.
2107 Handed over on a private session key for the same reason its two
2108 siblings are: the frame must not reach the bulk-export cache signature.
2109 """
2110 try:
2111 import streamlit as st
2113 return st.session_state.get("_export_text_metadata")
2114 except Exception: # no script run context (API, CLI)
2115 return None
2118def _session_trial_metadata():
2119 """The attached trial table, when running inside the app (DATA-29).
2121 Handed over on a private session key for the same reason its participant
2122 twin is: the frame must not reach the bulk-export cache signature.
2123 """
2124 try:
2125 import streamlit as st
2127 return st.session_state.get("_export_trial_metadata")
2128 except Exception: # no script run context (API, CLI)
2129 return None
2132def _session_participant_metadata():
2133 """The attached participant table, when running inside the app (DATA-20).
2135 Handed over through session state rather than through ``settings`` so the
2136 frame never reaches the bulk-export cache signature, which stringifies the
2137 settings dict and would truncate a long table into a colliding key. Returns
2138 ``None`` headlessly, where callers pass the frame in ``settings`` instead.
2139 """
2140 try:
2141 import streamlit as st
2143 return st.session_state.get("_export_participant_metadata")
2144 except Exception: # no script run context (API, CLI)
2145 return None
2148#: `index.csv`'s columns, in order.
2149INVENTORY_COLUMNS = (
2150 "path",
2151 "artifact",
2152 "format",
2153 "participant_id",
2154 "trial_id",
2155 "screen_id",
2156 "status",
2157 "note",
2158)
2161def _write_inventory(zf: zipfile.ZipFile, inventory: list[dict]) -> None:
2162 """Write ``index.csv``: one row per file in the bundle, plus each requested
2163 file that failed and each reading skipped. ``status`` is ``written``,
2164 ``failed`` or ``skipped``; a file type nobody asked for has no row."""
2165 frame = pd.DataFrame(inventory, columns=list(INVENTORY_COLUMNS))
2166 zf.writestr("index.csv", frame.to_csv(index=False))
2169def screen_choices(*frames: pd.DataFrame | None) -> list[str]:
2170 """The screen ids ``frames`` carry, for the export's screen picker: in the
2171 order they are shown (their lowest ``screen_index``), then by id. Empty when
2172 no frame has screens."""
2173 parts = [
2174 # De-duplicated first: a raw-gaze table can hold millions of samples.
2175 frame[
2176 [c for c in (SCREEN_ID, SCREEN_INDEX) if c in frame.columns]
2177 ].drop_duplicates()
2178 for frame in frames
2179 if frame is not None and not frame.empty and SCREEN_ID in frame.columns
2180 ]
2181 if not parts:
2182 return []
2183 pairs = pd.concat(parts, ignore_index=True).dropna(subset=[SCREEN_ID])
2184 pairs[SCREEN_ID] = pairs[SCREEN_ID].astype(str)
2185 if SCREEN_INDEX not in pairs.columns:
2186 pairs[SCREEN_INDEX] = float("nan")
2187 pairs[SCREEN_INDEX] = pd.to_numeric(pairs[SCREEN_INDEX], errors="coerce")
2188 first = pairs.groupby(SCREEN_ID)[SCREEN_INDEX].min().reset_index()
2189 first = first.sort_values([SCREEN_INDEX, SCREEN_ID], na_position="last")
2190 return first[SCREEN_ID].tolist()
2193def export_units(
2194 combos: pd.DataFrame,
2195 words: pd.DataFrame,
2196 fixations: pd.DataFrame,
2197 raw_gaze: pd.DataFrame | None = None,
2198 screens: tuple[str, ...] | None = None,
2199) -> pd.DataFrame:
2200 """One row per **screen export unit**: each trial in ``combos``, or each of
2201 its screens when it is a multipart reading. ``combos`` is already scoped.
2203 ``screens`` (``ExportOptions.screens``) keeps only the units on those
2204 screen ids, so a trial without screens has none of them and is left out.
2205 """
2206 rows: list[dict] = []
2207 for combo in combos.to_dict("records"):
2208 participant, trial = combo["participant_id"], combo["trial_id"]
2209 parent_words = extract_trial(words, participant, trial)
2210 parent_fixations = extract_trial(fixations, participant, trial)
2211 catalog = part_catalog(parent_words, parent_fixations)
2212 if (
2213 catalog.empty
2214 and parent_words.empty
2215 and parent_fixations.empty
2216 and raw_gaze is not None
2217 and SCREEN_ID in raw_gaze.columns
2218 ):
2219 # VIZ-45, as `api._select_part` decides it: a trial recorded as raw
2220 # gaze alone takes its screens from its samples, so each screen's
2221 # coordinate space is exported on its own instead of pooled.
2222 catalog = part_catalog(extract_trial(raw_gaze, participant, trial))
2223 if catalog.empty:
2224 rows.append(combo)
2225 else:
2226 for screen in catalog.to_dict("records"):
2227 rows.append({**combo, **screen})
2228 units = pd.DataFrame(rows)
2229 if screens is None:
2230 return units
2231 if units.empty or SCREEN_ID not in units.columns:
2232 return units.iloc[0:0]
2233 chosen = {str(screen) for screen in screens}
2234 keep = units[SCREEN_ID].notna() & units[SCREEN_ID].astype(str).isin(chosen)
2235 return units[keep].reset_index(drop=True)
2238@dataclass(frozen=True)
2239class ExportPlan:
2240 """What a bundle will hold, worked out before it is built.
2242 ``layer_files`` is a ceiling: a screen writes one file per layer it
2243 actually draws, which is only known once its figure is built.
2244 """
2246 trials: int
2247 units: int
2248 figure_files: int
2249 layer_files: int
2252def apply_export_scope(combos: pd.DataFrame, options: ExportOptions) -> pd.DataFrame:
2253 """``combos`` narrowed to ``options``' scope, as :func:`bulk_export` does."""
2254 return _apply_scope(combos, options)
2257def plan_export(
2258 combos: pd.DataFrame,
2259 words: pd.DataFrame,
2260 fixations: pd.DataFrame,
2261 options: ExportOptions,
2262 raw_gaze: pd.DataFrame | None = None,
2263) -> ExportPlan:
2264 """The trial, screen and figure-file counts :func:`bulk_export` will
2265 produce for these inputs and ``options``."""
2266 scoped = _apply_scope(combos, options)
2267 trials, units = count_export(scoped, words, fixations, raw_gaze, options.screens)
2268 return plan_from_counts(trials, units, options)
2271def count_export_units(
2272 combos: pd.DataFrame,
2273 words: pd.DataFrame,
2274 fixations: pd.DataFrame,
2275 raw_gaze: pd.DataFrame | None = None,
2276) -> int:
2277 """``len(export_units(...))`` — without walking every trial when no frame
2278 carries screens, the common case, where each trial is one unit."""
2279 if not any(
2280 frame is not None and SCREEN_ID in frame.columns
2281 for frame in (words, fixations, raw_gaze)
2282 ):
2283 return len(combos)
2284 return len(export_units(combos, words, fixations, raw_gaze))
2287def count_export(
2288 combos: pd.DataFrame,
2289 words: pd.DataFrame,
2290 fixations: pd.DataFrame,
2291 raw_gaze: pd.DataFrame | None = None,
2292 screens: tuple[str, ...] | None = None,
2293) -> tuple[int, int]:
2294 """``(trials, units)`` a bundle of ``combos`` exports: with ``screens``
2295 chosen, only the trials that show one of them count."""
2296 if screens is None:
2297 return len(combos), count_export_units(combos, words, fixations, raw_gaze)
2298 units = export_units(combos, words, fixations, raw_gaze, screens)
2299 if units.empty:
2300 return 0, 0
2301 trials = units[["participant_id", "trial_id"]].drop_duplicates()
2302 return len(trials), len(units)
2305def _keep_unit_trials(combos: pd.DataFrame, units: pd.DataFrame) -> pd.DataFrame:
2306 """``combos`` cut to the trials ``units`` still holds — after a screen
2307 choice, the trials the bundle actually exports."""
2308 if units.empty:
2309 return combos.iloc[0:0]
2310 keys = set(
2311 zip(
2312 units["participant_id"].astype(str),
2313 units["trial_id"].astype(str),
2314 strict=True,
2315 )
2316 )
2317 held = [
2318 (str(pid), str(tid)) in keys
2319 for pid, tid in zip(combos["participant_id"], combos["trial_id"], strict=True)
2320 ]
2321 return combos[held]
2324def plan_from_counts(trials: int, units: int, options: ExportOptions) -> ExportPlan:
2325 """:class:`ExportPlan` for known trial and unit counts — the formats'
2326 multiplication, apart from the (slower) unit expansion."""
2327 return ExportPlan(
2328 trials=trials,
2329 units=units,
2330 figure_files=units * len(options.figure_formats()),
2331 layer_files=units * len(SCANPATH_LAYER_ORDER) * len(options.layer_formats()),
2332 )
2335def describe_plan(plan: ExportPlan) -> str:
2336 """One line for the panel: what Build export is about to write."""
2338 def plural(n: int, word: str) -> str:
2339 return f"{n:,} {word}{'' if n == 1 else 's'}"
2341 readings = plural(plan.trials, "trial")
2342 if plan.units != plan.trials:
2343 readings += f" ({plural(plan.units, 'screen')})"
2344 parts = []
2345 if plan.figure_files:
2346 parts.append(plural(plan.figure_files, "figure file"))
2347 if plan.layer_files:
2348 parts.append(f"up to {plural(plan.layer_files, 'layer file')}")
2349 if not parts:
2350 return f"Exports {readings}."
2351 return f"Exports {readings}: {' and '.join(parts)}."
2354def _package_version() -> str:
2355 from scanpath_studio import __version__
2357 return __version__
2360def _plural(n: int, word: str) -> str:
2361 return f"{n:,} {word}{'' if n == 1 else 's'}"
2364def _scope_lines(
2365 options: ExportOptions, combos: pd.DataFrame, units: pd.DataFrame
2366) -> list[str]:
2367 """The README's *Scope* section: which trials the bundle was built from."""
2368 if options.scope == "trial":
2369 chosen = (
2370 f"one trial (participant {options.scope_participant}, "
2371 f"trial {options.scope_trial})"
2372 )
2373 elif options.scope == "participant":
2374 chosen = f"one participant ({options.scope_participant})"
2375 elif options.scope == "text":
2376 chosen = f"one text ({options.scope_text})"
2377 elif options.export_unfiltered:
2378 chosen = "the whole dataset, ignoring the trial filters"
2379 else:
2380 chosen = "the trials passing the trial filters"
2381 lines = []
2382 if options.dataset_name:
2383 lines.append(f"- Dataset: {options.dataset_name}")
2384 lines += [
2385 f"- Trials: {chosen}",
2386 f"- {_plural(len(combos), 'trial')}"
2387 + (f" ({_plural(len(units), 'screen')})" if len(units) != len(combos) else ""),
2388 ]
2389 if options.screens is not None:
2390 lines.append(f"- Screens: only {', '.join(options.screens)}")
2391 return lines
2394def bulk_export(
2395 combos: pd.DataFrame,
2396 words: pd.DataFrame,
2397 fixations: pd.DataFrame,
2398 *,
2399 canvas_width: int,
2400 canvas_height: int,
2401 base_font_size: int,
2402 font_family: str,
2403 x_field: str,
2404 y_field: str,
2405 settings: dict,
2406 options: ExportOptions,
2407 raw_gaze: pd.DataFrame | None = None,
2408 progress_callback=None,
2409 status_callback: StatusCallback | None = None,
2410 metadata_rows_for=None,
2411 annotation_records: list[dict] | None = None,
2412 annotation_dataset: str | None = None,
2413 column_names: Mapping[str, ColumnNames] | None = None,
2414) -> tuple[bytes, ExportProgress]:
2415 """Build a zip archive of selected artifacts and return its bytes.
2417 ``column_names`` (DATA-66) is the dataset's map per table
2418 (``{"fixations": …, "words": …, "raw_gaze": …}``): each table is written
2419 with the columns its file named under those names, the README's data
2420 dictionary says where every column came from, and a ``columns.json``
2421 records the map; patterns accept either name. Without it (headless
2422 callers, until the API carries maps) the tables keep the internal names.
2424 ``metadata_rows_for(participant, trial, text_id)`` (EXP-22) returns a
2425 trial's metadata-table rows for ``{table.field}`` patterns — the app passes
2426 ``metadata.pattern_rows``; headless callers have no attached tables.
2428 ``annotation_records`` (UX-179) are the annotations to write as
2429 ``annotations.json`` when ``options.include_annotations`` is set, in
2430 ``annotations.current_records()``'s shape; only those on the exported
2431 trials go in. The app passes the session's; headless callers have none.
2432 ``annotation_dataset`` is the dataset they were made on, which the file
2433 names (annotations file schema 3, DATA-48).
2435 progress_callback (if given) is invoked with an ExportProgress after every
2436 trial so the UI can update a progress bar.
2438 Raises ``ValueError`` before any work when ``options.path_pattern`` is not
2439 a path that stays inside the ZIP (:func:`path_structure_error`).
2440 """
2441 pattern_problem = path_structure_error(options.path_pattern)
2442 if pattern_problem:
2443 raise ValueError(pattern_problem)
2444 combos = _apply_scope(combos, options)
2445 units = export_units(combos, words, fixations, raw_gaze, options.screens)
2446 if options.screens is not None:
2447 # Only the trials that show a chosen screen are exported — for the
2448 # annotations, the README and every per-trial table alike.
2449 combos = _keep_unit_trials(combos, units)
2450 # UX-179's annotations, cut to the exported trials once: the README says
2451 # the file is there exactly when the writer below writes it (round 10).
2452 annotations_kept: list[dict] = []
2453 if options.include_annotations and annotation_records:
2454 from .annotations import records_in, records_to_store
2456 trials = zip(combos["participant_id"], combos["trial_id"], strict=True)
2457 annotations_kept = records_in(records_to_store(annotation_records), trials)
2458 maps = {
2459 table: names
2460 for table, names in (column_names or {}).items()
2461 if names is not None and names.entries
2462 }
2463 every_table = across_tables(maps)
2464 # Each table's columns as written, decided once from the whole table so
2465 # every per-trial file of it has the same columns — and what `columns.json`
2466 # and the README then describe.
2467 whole = {
2468 table: shareable_frame(frame)
2469 for table, frame in (
2470 ("fixations", fixations),
2471 ("words", words),
2472 ("raw_gaze", raw_gaze),
2473 )
2474 if table in maps and frame is not None and not frame.empty
2475 }
2476 hidden = {
2477 table: maps[table].redundant_aliases(frame) for table, frame in whole.items()
2478 }
2480 def names_for(artifact: str) -> tuple[ColumnNames, set[str] | None]:
2481 """The map an artifact is written with, and the aliases it leaves out:
2482 its own table's for the three tables, only the ids for a derived one."""
2483 table = _ARTIFACT_TABLE.get(artifact)
2484 if table in maps:
2485 return maps[table], hidden.get(table)
2486 return every_table.identity(), None
2488 progress = ExportProgress(total_trials=len(units))
2489 started = perf_counter()
2490 emit_status(
2491 status_callback,
2492 ExportStage.PREPARING,
2493 "Preparing trials and export manifest…",
2494 started_at=started,
2495 )
2496 buf = io.BytesIO()
2497 zf = zipfile.ZipFile(buf, "w", compression=zipfile.ZIP_DEFLATED)
2498 # The bundle's inventory, written last as `index.csv`: every file in it at
2499 # its actual path, and every requested file that failed, with the reading
2500 # and screen it belongs to.
2501 inventory: list[dict] = []
2503 def _inventory(path, artifact, status, unit=None, note="", fmt=None):
2504 inventory.append(
2505 {
2506 "path": path,
2507 "artifact": artifact,
2508 "format": PurePosixPath(path).suffix.lstrip(".")
2509 if fmt is None
2510 else fmt,
2511 "participant_id": (unit or {}).get("participant_id", ""),
2512 "trial_id": (unit or {}).get("trial_id", ""),
2513 "screen_id": (unit or {}).get("screen_id", ""),
2514 "status": status,
2515 "note": note,
2516 }
2517 )
2519 # EXP-23: each table's per-trial frames, stacked into `aggregate/all_<table>`
2520 # when the tables are combined. The family's words + fixations are kept
2521 # either way, for the reader summary, which only exists across trials.
2522 combined: dict[str, list[pd.DataFrame]] = {}
2523 family_words: list[pd.DataFrame] = []
2524 family_fixations: list[pd.DataFrame] = []
2525 # AN-32 / EXP-23: the export writes the reading measures the dataset
2526 # brought and computes none — so with none, the word tables are left out
2527 # and the README says why.
2528 measures_wanted = options.include_measures or options.include_analysis_family
2529 brought = brought_reading_measures(words)
2530 # What the bundle's own tables are written as, for its README and its
2531 # `columns.json`: only the tables it writes.
2532 exported = {
2533 "fixations": options.include_fixations or options.include_analysis_family,
2534 "words": measures_wanted and bool(brought),
2535 "raw_gaze": options.include_raw_gaze,
2536 }
2537 written = {
2538 table: written_columns(frame, maps[table])
2539 for table, frame in whole.items()
2540 if exported[table]
2541 }
2542 measure_headers = {
2543 row["canonical"]: row["column"] for row in written.get("words", [])
2544 }
2546 readme_lines = [
2547 "# Bulk export",
2548 f"Generated: {_local_stamp()}",
2549 "",
2550 f"Made with: {CITATION['title']}",
2551 f"Tool authors: {CITATION['authors']}",
2552 f"Version: {_package_version()}",
2553 f"DOI: https://doi.org/{CITATION['doi']}",
2554 "",
2555 "## Scope",
2556 *_scope_lines(options, combos, units),
2557 "",
2558 "## Layout",
2559 "- `index.csv` lists every file in this bundle — its path, what it is, "
2560 "its participant, trial and screen — and every requested file that "
2561 "failed (`status` = `failed`) or reading skipped (`skipped`). A file "
2562 "type that was not requested is not listed.",
2563 "- `per_trial/<participant>__<trial>/` holds each trial's files, "
2564 "unless the File naming pattern moved them (`index.csv` has every path).",
2565 "- A trial shown on several screens adds `screens/screen-001-<id>/` "
2566 "inside its folder.",
2567 *(
2568 [
2569 "- `aggregate/` holds each table once, every trial in this run "
2570 "stacked in it (*Combine all trials into one file*)."
2571 ]
2572 if options.combine_trials and options.any_table()
2573 else (
2574 ["- `aggregate/` holds the reader summary, across every trial."]
2575 if options.include_analysis_family
2576 else []
2577 )
2578 ),
2579 *(
2580 [
2581 "- `annotations.json` holds the favorites, tags and notes on "
2582 "these trials; import it on the app's Data Management page → Annotations."
2583 ]
2584 if annotations_kept
2585 else []
2586 ),
2587 "",
2588 "## Data dictionary",
2589 *(
2590 [
2591 "Columns keep the names they have in the dataset's own files. "
2592 "A column Scanpath Studio built, converted or computed keeps its "
2593 "internal name; `columns.json` maps every column below to its "
2594 "internal name, for scripts that work across datasets. Columns "
2595 "not listed are written under their own names. A table the app "
2596 "derives (a summary, the saccades) carries the dataset's ids under "
2597 "these names, and its own values under the app's.",
2598 *dictionary_lines(written, maps),
2599 *(
2600 ["", "Reading measures in the word tables:"]
2601 if measures_wanted
2602 else []
2603 ),
2604 ]
2605 if any(written.values())
2606 else [
2607 "Column names (Scanpath Studio's standard names):",
2608 "- participant_id, trial_id, text_id, word_id",
2609 "- screen_id, screen_index (multipart trials only)",
2610 "- x, y, width, height (word bounding boxes in screen px)",
2611 "- x, y, duration_ms, timestamp_ms (fixations)",
2612 ]
2613 ),
2614 *(
2615 [f"- {measure_headers.get(column, column)}" for column in brought]
2616 if brought
2617 else ["- (no reading measures — see below)"]
2618 if measures_wanted
2619 else []
2620 ),
2621 "",
2622 "## Reading measures",
2623 "The word tables carry the reading measures the dataset brought, as "
2624 "mapped on the app's Data Management page; Scanpath Studio computes none of them.",
2625 *(
2626 [
2627 "",
2628 "This dataset brought none, so the bundle has no word-measure "
2629 "table. Map them in the app on Data Management → Edit dataset → "
2630 "Reading measures.",
2631 ]
2632 if measures_wanted and not brought
2633 else []
2634 ),
2635 *(
2636 ["", f"Demo data note: {CITATION['corpus_note']}"]
2637 if options.dataset_name == DEMO_CHOICE
2638 else []
2639 ),
2640 ]
2641 zf.writestr("README.md", "\n".join(readme_lines))
2642 _inventory("README.md", "readme", "written")
2643 if any(written.values()):
2644 zf.writestr("columns.json", json.dumps(columns_manifest(written), indent=2))
2645 _inventory("columns.json", "columns", "written")
2646 if options.include_analysis_family:
2647 zf.writestr(
2648 "run_config.json",
2649 json.dumps(
2650 {
2651 "generated_at": datetime.now(UTC).isoformat(),
2652 "settings": settings,
2653 "preprocessing": settings.get("preprocessing", {}),
2654 },
2655 indent=2,
2656 default=str,
2657 ),
2658 )
2659 _inventory("run_config.json", "run_config", "written")
2661 # One warm Kaleido browser for every trial's figure (see _figure_renderer)
2662 # instead of cold-starting Chrome on each render. HTML needs no browser, so
2663 # only spin Kaleido up when a raster/vector format was requested (combined
2664 # figure or the per-layer breakdown).
2665 figure_formats = options.figure_formats()
2666 layer_formats = options.layer_formats()
2667 # EXP-1: a user pattern can map two trials to the same path (one that omits
2668 # the trial id, say). Two zip entries at one name silently loses a file, so
2669 # `resolve_export_path` disambiguates against what's already been written.
2670 used_paths: set = set()
2671 # The readers, trials and texts the bundle actually holds — what its
2672 # metadata tables are narrowed to at the end.
2673 exported_pairs: set[tuple[str, str]] = set()
2674 exported_texts: set[str] = set()
2675 emit_status(
2676 status_callback,
2677 (
2678 ExportStage.STARTING_RENDERER
2679 if options.needs_kaleido()
2680 else ExportStage.ENCODING_WRITING
2681 ),
2682 (
2683 "Starting one shared Chrome/Kaleido renderer…"
2684 if options.needs_kaleido()
2685 else "Writing selected files…"
2686 ),
2687 started_at=started,
2688 )
2689 with _figure_renderer(options.needs_kaleido()) as render_figure:
2690 for combo in units.itertuples(index=False):
2691 # The cancel checkpoint (`progress.Cancelled`), between units so a
2692 # stopped build never leaves one half-written; a no-op headlessly.
2693 try:
2694 report_progress(
2695 progress.finished_trials, progress.total_trials, unit="screens"
2696 )
2697 except BaseException: # `progress.Cancelled`: close the zip, go
2698 zf.close()
2699 raise
2700 participant = combo.participant_id
2701 trial = combo.trial_id
2702 screen_id = getattr(combo, SCREEN_ID, None)
2703 screen_index = getattr(combo, SCREEN_INDEX, None)
2704 screen_slug = (
2705 f"screen-{int(screen_index):03d}-{_safe_id(screen_id)}"
2706 if screen_id is not None and pd.notna(screen_id)
2707 else ""
2708 )
2709 slug = f"{_safe_id(participant)}__{_safe_id(trial)}"
2710 if screen_slug:
2711 slug += f"__{screen_slug}"
2712 unit_ids = {
2713 "participant_id": str(participant),
2714 "trial_id": str(trial),
2715 "screen_id": str(screen_id) if screen_slug else "",
2716 }
2718 # Slice via the same str-normalized position index the live view uses
2719 # (utils.extract_trial), so the export selects *exactly* what the trial
2720 # picker shows — not a raw dtype-sensitive boolean mask that can silently
2721 # miss rows the view finds.
2722 trial_words = extract_trial(words, participant, trial)
2723 trial_fix = extract_trial(fixations, participant, trial)
2724 trial_raw_gaze = (
2725 extract_trial(raw_gaze, participant, trial)
2726 if raw_gaze is not None and not raw_gaze.empty
2727 else pd.DataFrame()
2728 )
2729 if screen_slug:
2730 trial_words = extract_part(
2731 trial_words, participant, trial, str(screen_id)
2732 )
2733 trial_fix = extract_part(trial_fix, participant, trial, str(screen_id))
2734 trial_raw_gaze = (
2735 extract_part(trial_raw_gaze, participant, trial, str(screen_id))
2736 if not trial_raw_gaze.empty and SCREEN_ID in trial_raw_gaze.columns
2737 else pd.DataFrame()
2738 )
2740 # Skip only a genuinely empty trial (nothing to draw). The figure
2741 # builder renders from fixations alone (words optional — boxes/labels)
2742 # or from words alone (AOI layout), and the live view does too; so a
2743 # words-that-don't-join / fixations-only trial must still export
2744 # instead of being skipped with "empty data" (VIZ-5). VIZ-45: and
2745 # a trial recorded as raw gaze alone is drawn from its samples.
2746 if trial_words.empty and trial_fix.empty and trial_raw_gaze.empty:
2747 progress.finished_trials += 1
2748 progress.trials_skipped += 1
2749 progress.errors.append(f"{slug}: empty data, skipped")
2750 _inventory("", "reading", "skipped", unit_ids, "no data", fmt="")
2751 if progress_callback:
2752 progress_callback(progress)
2753 continue
2755 exported_pairs.add((str(participant), str(trial)))
2756 exported_texts.update(
2757 _unit_text_ids(combo._asdict(), trial_words, trial_fix, trial_raw_gaze)
2758 )
2760 # A screen's own canvas, for its figure and its plot config alike.
2761 unit_canvas = (
2762 screen_canvas_size(trial_words)
2763 or screen_canvas_size(trial_fix)
2764 or screen_canvas_size(trial_raw_gaze)
2765 )
2766 unit_width = int(unit_canvas[0] if unit_canvas else canvas_width)
2767 unit_height = int(unit_canvas[1] if unit_canvas else canvas_height)
2769 # EXP-1/EXP-2: everything this trial's paths, title and caption can
2770 # substitute — its combo row plus counts and the settings summary.
2771 fields = pattern_fields(
2772 participant,
2773 trial,
2774 trial_words,
2775 trial_fix,
2776 settings,
2777 combo_row=combo._asdict(),
2778 dataset_name=options.dataset_name,
2779 column_names=every_table,
2780 metadata_rows=(
2781 metadata_rows_for(
2782 participant, trial, combo._asdict().get("text_id")
2783 )
2784 if metadata_rows_for is not None
2785 else None
2786 ),
2787 )
2789 def _path(
2790 artifact: str, ext: str, _f=fields, _slug=screen_slug, reserve=True
2791 ) -> str:
2792 # `reserve=False` names a file that failed: the path it would
2793 # have had, for the inventory, without taking it from the next.
2794 if _slug:
2795 artifact = f"screens/{_slug}/{artifact}"
2796 return resolve_export_path(
2797 options.path_pattern,
2798 _f,
2799 artifact=artifact,
2800 ext=ext,
2801 used=used_paths if reserve else set(used_paths),
2802 )
2804 title = (
2805 render_pattern(options.title_pattern, fields)
2806 if options.title_pattern
2807 else ""
2808 )
2809 caption = (
2810 render_pattern(options.caption_pattern, fields)
2811 if options.caption_pattern
2812 else ""
2813 )
2815 if options.needs_figure():
2816 emit_status(
2817 status_callback,
2818 ExportStage.RASTERIZING,
2819 f"Rendering {progress.finished_trials + 1} of {progress.total_trials}…",
2820 started_at=started,
2821 completed=progress.finished_trials,
2822 total=progress.total_trials,
2823 )
2824 fig = None
2825 try:
2826 # EXP-4 / VIZ-24: apply the PRE-3 drift correction to the
2827 # figure's fixations (a no-op when "Off"), so the batch
2828 # matches the corrected figure on screen. `trial_fix` itself
2829 # stays uncorrected — the tables below export the recording.
2830 fig_fix, connector_y = _drift_corrected_for_figure(
2831 trial_fix, trial_words, settings
2832 )
2833 render_values = {
2834 name: settings[name]
2835 for name in STATIC_FIGURE_OPTIONS
2836 if name in settings
2837 }
2838 render_settings = FigureSettings.from_mapping(
2839 render_values,
2840 canvas_width=unit_width,
2841 canvas_height=unit_height,
2842 base_font_size=int(base_font_size),
2843 font_family=font_family,
2844 x_field=x_field,
2845 y_field=y_field,
2846 show_connectors=connector_y is not None,
2847 connector_y=connector_y,
2848 # A corrected figure colours by line, exactly as the
2849 # on-screen static path forces it (tabs.py PRE-3
2850 # overrides) — else the batch differs in colouring.
2851 color_by_line=bool(settings.get("color_by_line", False))
2852 or fig_fix is not trial_fix,
2853 )
2854 # VIZ-45: the trial's own samples, as the live figure
2855 # gets them (the layer self-gates on `show_raw_gaze`).
2856 fig = make_scanpath_figure(
2857 trial_words,
2858 fig_fix,
2859 settings=render_settings,
2860 raw_gaze=trial_raw_gaze if not trial_raw_gaze.empty else None,
2861 )
2862 # EXP-2: stamp the title/caption BEFORE measuring the output
2863 # size — the bands grow the figure, and rendering at the
2864 # pre-title size would crop them off.
2865 annotate_figure(fig, title=title, caption=caption)
2866 except Exception as exc:
2867 fig = None
2868 # Its formats, and its layer set when one was asked for.
2869 progress.figures_failed += len(figure_formats) + bool(layer_formats)
2870 progress.errors.append(f"{slug}: figure export failed ({exc})")
2871 for fmt in figure_formats:
2872 _inventory(
2873 _path("figure", fmt, reserve=False),
2874 "figure",
2875 "failed",
2876 unit_ids,
2877 str(exc),
2878 )
2879 if layer_formats:
2880 _inventory("", "layers", "failed", unit_ids, str(exc), fmt="")
2881 # EXP-24: each format on its own, so a missing browser costs the
2882 # PNG/SVG/PDF and still leaves the trial's HTML in the zip.
2883 if fig is not None:
2884 # Render at the figure's own fitted size (not the raw
2885 # monitor canvas) so the exported reading text matches the
2886 # on-screen scale.
2887 out_w = int(fig.layout.width or unit_width)
2888 out_h = int(fig.layout.height or unit_height)
2889 for fmt in figure_formats:
2890 try:
2891 if fmt == "html":
2892 # Browser-free + interactive; no Kaleido needed.
2893 data = fig.to_html(
2894 include_plotlyjs=html_plotlyjs(
2895 options.html_self_contained
2896 ),
2897 full_html=True,
2898 config={**PLOTLY_CONFIG},
2899 ).encode("utf-8")
2900 else:
2901 scale = options.png_scale if fmt == "png" else 1
2902 data = render_figure(fig, fmt, out_w, out_h, scale)
2903 path = _path("figure", fmt)
2904 zf.writestr(path, data)
2905 except Exception as exc:
2906 progress.figures_failed += 1
2907 progress.errors.append(
2908 f"{slug}: {fmt.upper()} figure export failed ({exc})"
2909 )
2910 _inventory(
2911 _path("figure", fmt, reserve=False),
2912 "figure",
2913 "failed",
2914 unit_ids,
2915 str(exc),
2916 )
2917 continue
2918 _inventory(path, "figure", "written", unit_ids)
2919 progress.bytes_written += len(data)
2920 progress.figures_written += 1
2922 # VIZ-5: per-layer breakdown into `layers/<layer>.<fmt>` — each a
2923 # copy of the figure with only one layer's elements, same
2924 # size/ranges so they register when stacked in Illustrator. Kept in
2925 # its own try so a layer-render failure is reported as such and
2926 # doesn't get misattributed to the combined figure (which may have
2927 # already been written above).
2928 if layer_formats and fig is not None:
2929 try:
2930 out_w = int(fig.layout.width or unit_width)
2931 out_h = int(fig.layout.height or unit_height)
2932 for layer_name, layer_fig in split_scanpath_layers(fig).items():
2933 for fmt in layer_formats:
2934 scale = options.png_scale if fmt == "png" else 1
2935 data = render_figure(
2936 layer_fig, fmt, out_w, out_h, scale
2937 )
2938 path = _path(f"layers/{layer_name}", fmt)
2939 zf.writestr(path, data)
2940 _inventory(
2941 path, f"layer:{layer_name}", "written", unit_ids
2942 )
2943 progress.bytes_written += len(data)
2944 progress.figures_written += 1
2945 except Exception as exc:
2946 progress.figures_failed += 1
2947 progress.errors.append(f"{slug}: layer export failed ({exc})")
2948 _inventory("", "layers", "failed", unit_ids, str(exc), fmt="")
2950 if options.include_plot_config:
2951 # Same guard as _drift_corrected_for_figure: correction runs
2952 # when an algorithm is set and the trial has both frames.
2953 algorithm = settings.get("align_algorithm")
2954 drift_applied = (
2955 bool(algorithm)
2956 and str(algorithm) != "Off"
2957 and not trial_fix.empty
2958 and not trial_words.empty
2959 )
2960 cfg = _plot_config_dict(
2961 participant,
2962 trial,
2963 unit_width,
2964 unit_height,
2965 x_field,
2966 y_field,
2967 settings,
2968 screen_id=str(screen_id) if screen_slug else None,
2969 drift_applied=drift_applied,
2970 )
2971 # EXP-2: the title/caption are part of how the figure looked, so
2972 # the manifest records them verbatim alongside the settings.
2973 cfg["figure_text"] = {"title": title, "caption": caption}
2974 # VIZ-50: the trial's samples, and whether they were recorded —
2975 # the bundled demo's are synthesized, which its figure and its
2976 # raw-gaze table cannot say for themselves.
2977 if not trial_raw_gaze.empty:
2978 cfg["raw_gaze"] = {
2979 "points": len(trial_raw_gaze),
2980 "synthesized": bool(settings.get("raw_gaze_synthesized")),
2981 }
2982 data = json.dumps(cfg, indent=2).encode("utf-8")
2983 path = _path("plot_config", "json")
2984 zf.writestr(path, data)
2985 _inventory(path, "plot_config", "written", unit_ids)
2986 progress.bytes_written += len(data)
2988 # AN-32 / EXP-23: the word table *is* the measures table — it carries
2989 # what the dataset brought, and nothing is computed. A trial with no
2990 # words, or a dataset that brought no measures, writes none.
2991 per_trial_measures = (
2992 trial_words
2993 if measures_wanted and brought and not trial_words.empty
2994 else None
2995 )
2996 family = {}
2997 if options.include_analysis_family:
2998 measured = trial_words
2999 analysis_fix = (
3000 enrich_fixations(
3001 assign_fixations_to_words(trial_fix, trial_words), trial_words
3002 )
3003 if not trial_fix.empty and not trial_words.empty
3004 else trial_fix
3005 )
3006 family = {
3007 "fixations": analysis_fix,
3008 "word_measures": per_trial_measures,
3009 "saccades": saccade_table(
3010 analysis_fix,
3011 pixels_per_degree=settings.get("pixels_per_degree"),
3012 raw_gaze=trial_raw_gaze,
3013 words=trial_words,
3014 ),
3015 "sentence_measures": sentence_measures(measured, analysis_fix),
3016 "trial_summary": trial_summary_table(measured, analysis_fix),
3017 "characters": character_grid(trial_words),
3018 "cleaning_qa": cleaning_report(
3019 analysis_fix,
3020 short_policy=(settings.get("preprocessing") or {}).get(
3021 "short_policy", "Off"
3022 ),
3023 ),
3024 }
3026 # The family's word-enriched fixations and its word_measures stand
3027 # in for the plain Fixations / Word measures files (EXP-15: two
3028 # members with one name, and a reader keeps only one of them).
3029 tables = {
3030 "fixations": trial_fix
3031 if options.include_fixations and not options.include_analysis_family
3032 else None,
3033 "raw_gaze": trial_raw_gaze if options.include_raw_gaze else None,
3034 "measures": per_trial_measures
3035 if options.include_measures and not options.include_analysis_family
3036 else None,
3037 **family,
3038 }
3039 tables = {
3040 artifact: table
3041 for artifact, table in tables.items()
3042 if table is not None and not table.empty
3043 }
3044 if options.combine_trials:
3045 for artifact, table in tables.items():
3046 combined.setdefault(artifact, []).append(table)
3047 else:
3048 for fmt in options.table_formats():
3049 for artifact, table in tables.items():
3050 path = _path(artifact, fmt)
3051 progress.bytes_written += _write_table(
3052 zf, path, table, fmt, *names_for(artifact)
3053 )
3054 _inventory(path, artifact, "written", unit_ids)
3055 if options.include_analysis_family:
3056 family_words.append(measured)
3057 family_fixations.append(family["fixations"])
3059 progress.finished_trials += 1
3060 if progress_callback:
3061 progress_callback(progress)
3062 emit_status(
3063 status_callback,
3064 ExportStage.ENCODING_WRITING,
3065 f"Wrote {progress.finished_trials} of {progress.total_trials}…",
3066 started_at=started,
3067 completed=progress.finished_trials,
3068 total=progress.total_trials,
3069 )
3071 # A reader summary has no per-trial form, so it is written across the
3072 # exported trials whether or not the tables are combined.
3073 if options.include_analysis_family and family_words:
3074 summary = reader_summary_table(
3075 pd.concat(family_words, ignore_index=True),
3076 pd.concat(family_fixations, ignore_index=True),
3077 )
3078 if not summary.empty:
3079 combined["reader_summary"] = [summary]
3080 stacked = {
3081 artifact: pd.concat(frames, ignore_index=True)
3082 for artifact, frames in combined.items()
3083 }
3084 for fmt in options.table_formats():
3085 for artifact, table in stacked.items():
3086 path = f"aggregate/all_{artifact}.{fmt}"
3087 progress.bytes_written += _write_table(
3088 zf, path, table, fmt, *names_for(artifact)
3089 )
3090 _inventory(path, artifact, "written")
3091 # DATA-20: the participant table travels as its own per-grain table rather
3092 # than as columns smeared across the trial files — which is what keeps a
3093 # reader attribute distinguishable from a per-fixation measurement on the
3094 # way out, exactly as it is on the way in. Whenever one is attached, narrowed
3095 # to `options.metadata_fields` (milestone 10's per-field opt-out; `None` is
3096 # every field). The frame is handed over out-of-band (the settings dict
3097 # carries only its fingerprint, so the bulk-export cache signature stays
3098 # honest — see the note in `tabs._render_export_panel`). API/CLI callers can
3099 # pass the frame in `settings` directly, which still works.
3100 participant_metadata = (settings or {}).get("participant_metadata")
3101 if participant_metadata is None:
3102 participant_metadata = _session_participant_metadata()
3103 participant_metadata = _rows_in_scope(
3104 _selected_metadata_columns(participant_metadata, options.metadata_fields),
3105 pairs=exported_pairs,
3106 texts=exported_texts,
3107 grain="participant",
3108 )
3109 if participant_metadata is not None and not participant_metadata.empty:
3110 for fmt in options.table_formats():
3111 path = f"metadata/participants.{fmt}"
3112 progress.bytes_written += _write_table(zf, path, participant_metadata, fmt)
3113 _inventory(path, "participant_metadata", "written")
3114 # DATA-29: and the trial table beside it, on the same terms — its own
3115 # per-grain file, keyed as it was attached, so a reading's attributes stay
3116 # distinguishable from the per-fixation measurements of that reading.
3117 trial_metadata = (settings or {}).get("trial_metadata")
3118 if trial_metadata is None:
3119 trial_metadata = _session_trial_metadata()
3120 trial_metadata = _rows_in_scope(
3121 _selected_trial_metadata_columns(trial_metadata, options.trial_metadata_fields),
3122 pairs=exported_pairs,
3123 texts=exported_texts,
3124 grain="trial",
3125 )
3126 if trial_metadata is not None and not trial_metadata.empty:
3127 for fmt in options.table_formats():
3128 path = f"metadata/trials.{fmt}"
3129 progress.bytes_written += _write_table(zf, path, trial_metadata, fmt)
3130 _inventory(path, "trial_metadata", "written")
3131 # And the text table, the third grain — same reasoning again.
3132 text_metadata = (settings or {}).get("text_metadata")
3133 if text_metadata is None:
3134 text_metadata = _session_text_metadata()
3135 text_metadata = _rows_in_scope(
3136 _selected_text_metadata_columns(text_metadata, options.text_metadata_fields),
3137 pairs=exported_pairs,
3138 texts=exported_texts,
3139 grain="text",
3140 )
3141 if text_metadata is not None and not text_metadata.empty:
3142 for fmt in options.table_formats():
3143 path = f"metadata/texts.{fmt}"
3144 progress.bytes_written += _write_table(zf, path, text_metadata, fmt)
3145 _inventory(path, "text_metadata", "written")
3146 # UX-179: the exported trials' annotations, in the Data → Annotations file
3147 # format, so the bundle's notes can be imported back into the app.
3148 if annotations_kept:
3149 from .annotations import records_to_store, serialize
3151 data = serialize(
3152 records_to_store(annotations_kept), dataset=annotation_dataset
3153 ).encode("utf-8")
3154 zf.writestr("annotations.json", data)
3155 _inventory("annotations.json", "annotations", "written")
3156 progress.bytes_written += len(data)
3157 emit_status(
3158 status_callback,
3159 ExportStage.FINALIZING,
3160 "Finalizing and compressing the zip archive…",
3161 started_at=started,
3162 completed=progress.finished_trials,
3163 total=progress.total_trials,
3164 )
3165 _write_inventory(zf, inventory)
3166 progress.files_written = len(zf.namelist())
3167 zf.close()
3168 buf.seek(0)
3169 result = buf.getvalue()
3170 emit_status(
3171 status_callback,
3172 ExportStage.READY,
3173 "Export archive is ready.",
3174 started_at=started,
3175 completed=progress.finished_trials,
3176 total=progress.total_trials,
3177 )
3178 return result, progress