Coverage for scanpath_studio/authoring.py: 91%
261 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Pure helpers for hand-authored scanpath documents and edits (VIZ-33)."""
3from __future__ import annotations
5import json
6import math
7import re
8from collections.abc import Mapping
9from dataclasses import dataclass
10from typing import Any
12import pandas as pd
14AUTHORING_SCHEMA = 2
15DEFAULT_LAYOUT = {
16 "canvas_width": 1200,
17 "margin": 70,
18 "character_width": 13,
19 "word_height": 32,
20 "line_height": 58,
21}
22EVENT_COLUMNS = [
23 "fixation_id",
24 "order_in_trial",
25 "word_id",
26 "x",
27 "y",
28 "duration_ms",
29]
32@dataclass(frozen=True)
33class AuthoringDocument:
34 """Portable source text, layout settings, and stable fixation events."""
36 text: str
37 events: pd.DataFrame
38 layout: dict[str, int]
39 schema: int = AUTHORING_SCHEMA
42def layout_text(
43 text: str,
44 *,
45 canvas_width: int = 1200,
46 margin: int = 70,
47 character_width: int = 13,
48 word_height: int = 32,
49 line_height: int = 58,
50) -> pd.DataFrame:
51 """Lay text into deterministic boxes while preserving explicit line breaks.
53 Source lines are handled independently. Long lines wrap at ``canvas_width``;
54 explicit blank lines still consume a line, so ``line_idx`` mirrors the text a
55 user typed instead of collapsing all whitespace into one paragraph.
56 """
57 rows: list[dict[str, Any]] = []
58 x, y, line = margin, margin, 0
59 word_index = 1
60 source_lines = str(text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
61 for source_index, source_line in enumerate(source_lines):
62 if source_index:
63 x, y, line = margin, y + line_height, line + 1
64 for token in re.findall(r"\S+", source_line):
65 width = max(character_width * len(token), character_width * 2)
66 if x > margin and x + width > canvas_width - margin:
67 x, y, line = margin, y + line_height, line + 1
68 rows.append(
69 {
70 "participant_id": "author",
71 "trial_id": "authored-1",
72 "text_id": "authored-text",
73 "word_id": float(word_index),
74 "text": token,
75 "line_idx": float(line),
76 # Canonical word geometry is box top-left + width/height.
77 "x": float(x),
78 "y": float(y),
79 "width": float(width),
80 "height": float(word_height),
81 }
82 )
83 word_index += 1
84 x += width + character_width
85 return pd.DataFrame(
86 rows,
87 columns=[
88 "participant_id",
89 "trial_id",
90 "text_id",
91 "word_id",
92 "text",
93 "line_idx",
94 "x",
95 "y",
96 "width",
97 "height",
98 ],
99 )
102def layout_problems(
103 words: pd.DataFrame, *, canvas_width: int, margin: int
104) -> list[str]:
105 """Describe geometry that cannot fit inside the configured horizontal bounds."""
106 if words.empty:
107 return []
108 overflow = words[
109 (words["x"] < margin) | (words["x"] + words["width"] > canvas_width - margin)
110 ]
111 if overflow.empty:
112 return []
113 ids = ", ".join(str(int(value)) for value in overflow["word_id"].head(8))
114 suffix = f" (+{len(overflow) - 8} more)" if len(overflow) > 8 else ""
115 return [f"Too wide for the canvas: word {ids}{suffix}."]
118def default_events(words: pd.DataFrame) -> pd.DataFrame:
119 """Return one stable, centered fixation per word as the authoring seed."""
120 if words.empty:
121 return pd.DataFrame(columns=EVENT_COLUMNS)
122 ids = pd.Series(range(1, len(words) + 1), dtype="int64")
123 return pd.DataFrame(
124 {
125 "fixation_id": ids,
126 "order_in_trial": ids,
127 "word_id": words["word_id"].astype(int).reset_index(drop=True),
128 "x": (words["x"] + words["width"] / 2).reset_index(drop=True),
129 "y": (words["y"] + words["height"] / 2).reset_index(drop=True),
130 "duration_ms": 220,
131 },
132 columns=EVENT_COLUMNS,
133 )
136def _number(value: Any) -> float | None:
137 try:
138 number = float(value)
139 except (TypeError, ValueError):
140 return None
141 return number if math.isfinite(number) else None
144def _positive_int(value: Any) -> int | None:
145 number = _number(value)
146 if number is None or number <= 0 or not number.is_integer():
147 return None
148 return int(number)
151def event_target_word(event: Mapping[str, Any]) -> int | None:
152 """Return an optional target word; spatial X/Y remain independently valid."""
153 return _positive_int(event.get("word_id"))
156def normalize_event_table(events: pd.DataFrame | None) -> pd.DataFrame:
157 """Migrate/edit an event table into the stable schema-2 column contract."""
158 frame = pd.DataFrame() if events is None else pd.DataFrame(events).copy()
159 for column in EVENT_COLUMNS:
160 if column not in frame:
161 frame[column] = None
162 frame = frame[EVENT_COLUMNS].reset_index(drop=True)
164 used_ids = [
165 value for value in (_positive_int(v) for v in frame["fixation_id"]) if value
166 ]
167 next_id = max(used_ids, default=0) + 1
168 used_orders = [
169 value for value in (_positive_int(v) for v in frame["order_in_trial"]) if value
170 ]
171 next_order = max(used_orders, default=0) + 1
172 for index in frame.index:
173 if _positive_int(frame.at[index, "fixation_id"]) is None:
174 frame.at[index, "fixation_id"] = next_id
175 next_id += 1
176 if _positive_int(frame.at[index, "order_in_trial"]) is None:
177 frame.at[index, "order_in_trial"] = next_order
178 next_order += 1
179 return frame
182def _is_scalar_cell(value: Any) -> bool:
183 return value is None or (
184 isinstance(value, (str, int, float)) and not isinstance(value, bool)
185 )
188def event_records_frame(records: Any) -> pd.DataFrame:
189 """Build the normalized event table from untrusted JSON records.
191 ``records`` comes from a link or a file, so its shape is checked before
192 pandas sees it: a list of objects, each naming at least one event column,
193 with plain number/text/null values. Anything else raises ``ValueError``
194 with a reason, rather than a ``TypeError`` from deep inside pandas.
195 """
196 if not isinstance(records, list):
197 raise ValueError("Authored fixations must be a list of objects.")
198 for number, record in enumerate(records, start=1):
199 if not isinstance(record, dict):
200 raise ValueError(f"Authored fixation {number} is not an object.")
201 if not any(column in record for column in EVENT_COLUMNS):
202 raise ValueError(
203 f"Authored fixation {number} has none of: {', '.join(EVENT_COLUMNS)}."
204 )
205 bad = sorted(
206 str(key)
207 for key in EVENT_COLUMNS
208 if key in record and not _is_scalar_cell(record[key])
209 )
210 if bad:
211 raise ValueError(
212 f"Authored fixation {number} has a non-scalar {', '.join(bad)}."
213 )
214 frame, _ = reconcile_event_table(pd.DataFrame(records).reset_index(drop=True))
215 return frame
218def event_problems(words: pd.DataFrame, events: pd.DataFrame | None) -> list[str]:
219 """Return actionable structural problems without silently renumbering rows."""
220 frame = normalize_event_table(events)
221 problems: list[str] = []
222 ids = [_positive_int(value) for value in frame["fixation_id"]]
223 orders = [_positive_int(value) for value in frame["order_in_trial"]]
224 duplicate_ids = sorted({value for value in ids if value and ids.count(value) > 1})
225 duplicate_orders = sorted(
226 {value for value in orders if value and orders.count(value) > 1}
227 )
228 if duplicate_ids:
229 problems.append(
230 "Fixation id must be unique; duplicate: "
231 + ", ".join(str(value) for value in duplicate_ids)
232 + "."
233 )
234 if duplicate_orders:
235 problems.append(
236 "Order must be unique; duplicate: "
237 + ", ".join(str(value) for value in duplicate_orders)
238 + "."
239 )
240 unusable = unusable_event_rows(words, frame)
241 if unusable:
242 listed = ", ".join(str(number) for number in unusable[:10])
243 suffix = f" (+{len(unusable) - 10} more)" if len(unusable) > 10 else ""
244 problems.append(
245 f"Not drawn — no X/Y and no valid target word: row {listed}{suffix}."
246 )
247 return problems
250def _structural_problems(events: pd.DataFrame) -> list[str]:
251 """Problems that make stable id/order reconciliation ambiguous."""
252 return [
253 problem
254 for problem in event_problems(pd.DataFrame(), events)
255 if problem.startswith(("Fixation id", "Order"))
256 ]
259def reconcile_event_table(
260 events: pd.DataFrame | None, selected_fixation_id: int | None = None
261) -> tuple[pd.DataFrame, int | None]:
262 """Normalize table edits and preserve selection by stable fixation id."""
263 frame = normalize_event_table(events)
264 problems = _structural_problems(frame)
265 if problems:
266 raise ValueError(" ".join(problems))
267 ids = {_positive_int(value) for value in frame["fixation_id"]}
268 selected = selected_fixation_id if selected_fixation_id in ids else None
269 return frame, selected
272def apply_authoring_event(
273 events: pd.DataFrame | None,
274 event: Mapping[str, Any],
275 *,
276 selected_fixation_id: int | None = None,
277) -> tuple[pd.DataFrame, int | None]:
278 """Apply one compact canvas event keyed by stable ``fixation_id``."""
279 frame, selected = reconcile_event_table(events, selected_fixation_id)
280 action = str(event.get("type", ""))
281 fixation_id = _positive_int(event.get("fixation_id"))
282 if action == "add":
283 x, y = _number(event.get("x")), _number(event.get("y"))
284 if x is None or y is None:
285 raise ValueError("A new fixation needs finite X and Y coordinates.")
286 existing_ids = [_positive_int(value) or 0 for value in frame["fixation_id"]]
287 existing_orders = [
288 _positive_int(value) or 0 for value in frame["order_in_trial"]
289 ]
290 fixation_id = max(existing_ids, default=0) + 1
291 frame.loc[len(frame)] = {
292 "fixation_id": fixation_id,
293 "order_in_trial": max(existing_orders, default=0) + 1,
294 "word_id": None,
295 "x": x,
296 "y": y,
297 "duration_ms": 220,
298 }
299 selected = fixation_id
300 elif action in {"move", "update"}:
301 if fixation_id is None:
302 raise ValueError("The canvas edit has no valid fixation id.")
303 matches = frame.index[
304 frame["fixation_id"].map(_positive_int).eq(fixation_id)
305 ].tolist()
306 if not matches:
307 raise ValueError(f"Fixation {fixation_id} no longer exists.")
308 allowed = (
309 {"x", "y"}
310 if action == "move"
311 else {
312 "x",
313 "y",
314 "word_id",
315 "duration_ms",
316 "order_in_trial",
317 }
318 )
319 for field in allowed:
320 if field in event:
321 frame.at[matches[0], field] = event[field]
322 selected = fixation_id
323 elif action == "delete":
324 if fixation_id is not None:
325 frame = frame[
326 ~frame["fixation_id"].map(_positive_int).eq(fixation_id)
327 ].reset_index(drop=True)
328 selected = None if selected == fixation_id else selected
329 elif action == "select":
330 selected = fixation_id
331 else:
332 raise ValueError(f"Unknown authoring event: {action or '(blank)'}.")
333 return reconcile_event_table(frame, selected)
336def _word_key(text: Any) -> str:
337 """A word as a target names it: letters and digits, case folded — so a
338 punctuation or capitalisation fix leaves the target where it was."""
339 return "".join(char for char in str(text).casefold() if char.isalnum())
342def _target_keys(words: pd.DataFrame) -> dict[int, str]:
343 if words.empty:
344 return {}
345 return {
346 int(word_id): _word_key(text)
347 for word_id, text in zip(words["word_id"], words["text"], strict=True)
348 }
351def stale_target_words(
352 old_words: pd.DataFrame, new_words: pd.DataFrame, events: pd.DataFrame | None
353) -> dict[int, tuple[int, str]]:
354 """The fixations whose target word a text edit removed or replaced, as
355 ``{fixation_id: (word_id, the word it named)}``.
357 Editing the stimulus keeps every authored fixation as it is — X/Y, order,
358 duration and id. Only the optional target word can go out of date: its id
359 may no longer exist, or now name a different word. Those are reported, not
360 changed; a punctuation or capitalisation fix to the word is not a change.
361 :func:`unresolved_targets` says which are still out of date later on.
362 """
363 if events is None or events.empty:
364 return {}
365 before, after = _target_keys(old_words), _target_keys(new_words)
366 stale: dict[int, tuple[int, str]] = {}
367 for event in normalize_event_table(events).to_dict("records"):
368 word_id = event_target_word(event)
369 fixation_id = _positive_int(event.get("fixation_id"))
370 if word_id is None or fixation_id is None or word_id not in before:
371 continue
372 if after.get(word_id) != before[word_id]:
373 stale[fixation_id] = (word_id, before[word_id])
374 return stale
377def unresolved_targets(
378 stale: Mapping[int, tuple[int, str]], words: pd.DataFrame, events: pd.DataFrame
379) -> dict[int, int]:
380 """``{fixation_id: word_id}`` for the entries of :func:`stale_target_words`
381 that are still out of date: the fixation exists, still targets that word id,
382 and the text there is still not the word it named. Changing or clearing the
383 target, deleting the fixation or undoing the text edit resolves one."""
384 if not stale or events is None or events.empty:
385 return {}
386 current = _target_keys(words)
387 targets = {
388 _positive_int(event.get("fixation_id")): event_target_word(event)
389 for event in normalize_event_table(events).to_dict("records")
390 }
391 return {
392 fixation_id: word_id
393 for fixation_id, (word_id, named) in stale.items()
394 if targets.get(fixation_id) == word_id and current.get(word_id) != named
395 }
398def destructive_change(before: pd.DataFrame | None, after: pd.DataFrame | None) -> bool:
399 """Whether going from ``before`` to ``after`` lost authored work: a fixation
400 removed, or one that is still there moved, retimed, reordered or retargeted.
402 Adding a fixation loses nothing, so it is not one. This is what decides when
403 the authoring screen keeps the previous draft for **Restore previous draft**.
404 """
405 old = normalize_event_table(before)
406 new = normalize_event_table(after)
407 if old.empty:
408 return False
410 def keyed(frame: pd.DataFrame) -> dict[int, tuple]:
411 rows = {}
412 for event in frame.to_dict("records"):
413 fixation_id = _positive_int(event.get("fixation_id"))
414 rows[fixation_id] = (
415 _number(event.get("x")),
416 _number(event.get("y")),
417 _number(event.get("duration_ms")),
418 _positive_int(event.get("order_in_trial")),
419 event_target_word(event),
420 )
421 return rows
423 old_rows, new_rows = keyed(old), keyed(new)
424 return any(new_rows.get(key) != value for key, value in old_rows.items())
427def unusable_event_rows(words: pd.DataFrame, events: pd.DataFrame) -> list[int]:
428 """Return rows that have neither usable X/Y nor a valid word fallback."""
429 if events is None or events.empty:
430 return []
431 valid_words = (
432 set(words["word_id"].astype(int))
433 if not words.empty and "word_id" in words
434 else set()
435 )
436 unusable: list[int] = []
437 for number, event in enumerate(events.to_dict("records"), start=1):
438 has_xy = (
439 _number(event.get("x")) is not None and _number(event.get("y")) is not None
440 )
441 if not has_xy and event_target_word(event) not in valid_words:
442 unusable.append(number)
443 return unusable
446def authored_fixations(words: pd.DataFrame, events: pd.DataFrame) -> pd.DataFrame:
447 """Normalize authored events into the ordinary canonical fixation schema."""
448 columns = [
449 "participant_id",
450 "trial_id",
451 "text_id",
452 "x",
453 "y",
454 "duration_ms",
455 "timestamp_ms",
456 "fixation_id",
457 "word_id",
458 "order_in_trial",
459 ]
460 if events is None or events.empty:
461 return pd.DataFrame(columns=columns)
462 frame, _ = reconcile_event_table(events)
463 centers = pd.DataFrame(columns=["x", "y"])
464 if not words.empty:
465 centers = words.set_index(words["word_id"].astype(int))[
466 ["x", "y", "width", "height"]
467 ].copy()
468 centers["x"] = centers["x"] + centers["width"] / 2
469 centers["y"] = centers["y"] + centers["height"] / 2
471 frame = frame.assign(
472 _order=frame["order_in_trial"].map(_positive_int),
473 _row=range(len(frame)),
474 ).sort_values(["_order", "_row"], kind="stable")
475 rows: list[dict[str, Any]] = []
476 timestamp = 0.0
477 for event in frame.to_dict("records"):
478 word_id = event_target_word(event)
479 x, y = _number(event.get("x")), _number(event.get("y"))
480 if (x is None or y is None) and word_id in centers.index:
481 x, y = float(centers.loc[word_id, "x"]), float(centers.loc[word_id, "y"])
482 if x is None or y is None:
483 continue
484 duration = _number(event.get("duration_ms"))
485 duration = duration if duration is not None and duration > 0 else 220.0
486 order = _positive_int(event.get("order_in_trial"))
487 fixation_id = _positive_int(event.get("fixation_id"))
488 rows.append(
489 {
490 "participant_id": "author",
491 "trial_id": "authored-1",
492 "text_id": "authored-text",
493 "x": x,
494 "y": y,
495 "duration_ms": duration,
496 "timestamp_ms": timestamp,
497 "fixation_id": float(fixation_id),
498 "word_id": float(word_id) if word_id in centers.index else float("nan"),
499 "order_in_trial": int(order),
500 }
501 )
502 timestamp += duration
503 return pd.DataFrame(rows, columns=columns)
506def _json_records(events: pd.DataFrame) -> list[dict[str, Any]]:
507 records: list[dict[str, Any]] = []
508 for row in normalize_event_table(events).to_dict("records"):
509 records.append(
510 {
511 key: None
512 if value is None or (isinstance(value, float) and math.isnan(value))
513 else value
514 for key, value in row.items()
515 }
516 )
517 return records
520def authoring_json(
521 text: str,
522 events: pd.DataFrame,
523 *,
524 layout: Mapping[str, int] | None = None,
525) -> str:
526 """Serialize a schema-2 source document for portable save/restore."""
527 layout_value = {**DEFAULT_LAYOUT, **dict(layout or {})}
528 return json.dumps(
529 {
530 "schema": AUTHORING_SCHEMA,
531 "text": text,
532 "layout": layout_value,
533 "fixations": _json_records(events),
534 },
535 indent=2,
536 )
539#: Schema 1 or 2 is accepted; the user is told only what the file is not.
540_NOT_AUTHORING = "That file isn't an authoring file saved from this editor."
543def parse_authoring_document(payload: str) -> AuthoringDocument:
544 """Restore schema 2 or migrate a VIZ-20 schema-1 authoring document."""
545 try:
546 value = json.loads(payload)
547 except ValueError:
548 value = None
549 if not isinstance(value, dict):
550 raise ValueError(_NOT_AUTHORING)
551 schema = value.get("schema")
552 if schema not in {1, AUTHORING_SCHEMA} or not isinstance(value.get("text"), str):
553 raise ValueError(_NOT_AUTHORING)
554 events = value.get("fixations", [])
555 if not isinstance(events, list):
556 raise ValueError("The authoring file's fixations must be a list.")
557 frame = event_records_frame(events)
558 layout = value.get("layout", {}) if schema == AUTHORING_SCHEMA else {}
559 if not isinstance(layout, dict):
560 raise ValueError("The authoring file's layout must be an object.")
561 unknown = sorted(set(layout) - set(DEFAULT_LAYOUT))
562 if unknown:
563 raise ValueError(f"Unknown authoring layout option: {', '.join(unknown)}.")
564 try:
565 normalized_layout = {
566 key: int({**DEFAULT_LAYOUT, **layout}[key]) for key in DEFAULT_LAYOUT
567 }
568 except (TypeError, ValueError) as exc:
569 raise ValueError("Authoring layout values must be whole numbers.") from exc
570 return AuthoringDocument(value["text"], frame, normalized_layout)