Coverage for scanpath_studio/authoring.py: 91%

261 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Pure helpers for hand-authored scanpath documents and edits (VIZ-33).""" 

2 

3from __future__ import annotations 

4 

5import json 

6import math 

7import re 

8from collections.abc import Mapping 

9from dataclasses import dataclass 

10from typing import Any 

11 

12import pandas as pd 

13 

14AUTHORING_SCHEMA = 2 

15DEFAULT_LAYOUT = { 

16 "canvas_width": 1200, 

17 "margin": 70, 

18 "character_width": 13, 

19 "word_height": 32, 

20 "line_height": 58, 

21} 

22EVENT_COLUMNS = [ 

23 "fixation_id", 

24 "order_in_trial", 

25 "word_id", 

26 "x", 

27 "y", 

28 "duration_ms", 

29] 

30 

31 

32@dataclass(frozen=True) 

33class AuthoringDocument: 

34 """Portable source text, layout settings, and stable fixation events.""" 

35 

36 text: str 

37 events: pd.DataFrame 

38 layout: dict[str, int] 

39 schema: int = AUTHORING_SCHEMA 

40 

41 

42def layout_text( 

43 text: str, 

44 *, 

45 canvas_width: int = 1200, 

46 margin: int = 70, 

47 character_width: int = 13, 

48 word_height: int = 32, 

49 line_height: int = 58, 

50) -> pd.DataFrame: 

51 """Lay text into deterministic boxes while preserving explicit line breaks. 

52 

53 Source lines are handled independently. Long lines wrap at ``canvas_width``; 

54 explicit blank lines still consume a line, so ``line_idx`` mirrors the text a 

55 user typed instead of collapsing all whitespace into one paragraph. 

56 """ 

57 rows: list[dict[str, Any]] = [] 

58 x, y, line = margin, margin, 0 

59 word_index = 1 

60 source_lines = str(text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n") 

61 for source_index, source_line in enumerate(source_lines): 

62 if source_index: 

63 x, y, line = margin, y + line_height, line + 1 

64 for token in re.findall(r"\S+", source_line): 

65 width = max(character_width * len(token), character_width * 2) 

66 if x > margin and x + width > canvas_width - margin: 

67 x, y, line = margin, y + line_height, line + 1 

68 rows.append( 

69 { 

70 "participant_id": "author", 

71 "trial_id": "authored-1", 

72 "text_id": "authored-text", 

73 "word_id": float(word_index), 

74 "text": token, 

75 "line_idx": float(line), 

76 # Canonical word geometry is box top-left + width/height. 

77 "x": float(x), 

78 "y": float(y), 

79 "width": float(width), 

80 "height": float(word_height), 

81 } 

82 ) 

83 word_index += 1 

84 x += width + character_width 

85 return pd.DataFrame( 

86 rows, 

87 columns=[ 

88 "participant_id", 

89 "trial_id", 

90 "text_id", 

91 "word_id", 

92 "text", 

93 "line_idx", 

94 "x", 

95 "y", 

96 "width", 

97 "height", 

98 ], 

99 ) 

100 

101 

102def layout_problems( 

103 words: pd.DataFrame, *, canvas_width: int, margin: int 

104) -> list[str]: 

105 """Describe geometry that cannot fit inside the configured horizontal bounds.""" 

106 if words.empty: 

107 return [] 

108 overflow = words[ 

109 (words["x"] < margin) | (words["x"] + words["width"] > canvas_width - margin) 

110 ] 

111 if overflow.empty: 

112 return [] 

113 ids = ", ".join(str(int(value)) for value in overflow["word_id"].head(8)) 

114 suffix = f" (+{len(overflow) - 8} more)" if len(overflow) > 8 else "" 

115 return [f"Too wide for the canvas: word {ids}{suffix}."] 

116 

117 

118def default_events(words: pd.DataFrame) -> pd.DataFrame: 

119 """Return one stable, centered fixation per word as the authoring seed.""" 

120 if words.empty: 

121 return pd.DataFrame(columns=EVENT_COLUMNS) 

122 ids = pd.Series(range(1, len(words) + 1), dtype="int64") 

123 return pd.DataFrame( 

124 { 

125 "fixation_id": ids, 

126 "order_in_trial": ids, 

127 "word_id": words["word_id"].astype(int).reset_index(drop=True), 

128 "x": (words["x"] + words["width"] / 2).reset_index(drop=True), 

129 "y": (words["y"] + words["height"] / 2).reset_index(drop=True), 

130 "duration_ms": 220, 

131 }, 

132 columns=EVENT_COLUMNS, 

133 ) 

134 

135 

136def _number(value: Any) -> float | None: 

137 try: 

138 number = float(value) 

139 except (TypeError, ValueError): 

140 return None 

141 return number if math.isfinite(number) else None 

142 

143 

144def _positive_int(value: Any) -> int | None: 

145 number = _number(value) 

146 if number is None or number <= 0 or not number.is_integer(): 

147 return None 

148 return int(number) 

149 

150 

151def event_target_word(event: Mapping[str, Any]) -> int | None: 

152 """Return an optional target word; spatial X/Y remain independently valid.""" 

153 return _positive_int(event.get("word_id")) 

154 

155 

156def normalize_event_table(events: pd.DataFrame | None) -> pd.DataFrame: 

157 """Migrate/edit an event table into the stable schema-2 column contract.""" 

158 frame = pd.DataFrame() if events is None else pd.DataFrame(events).copy() 

159 for column in EVENT_COLUMNS: 

160 if column not in frame: 

161 frame[column] = None 

162 frame = frame[EVENT_COLUMNS].reset_index(drop=True) 

163 

164 used_ids = [ 

165 value for value in (_positive_int(v) for v in frame["fixation_id"]) if value 

166 ] 

167 next_id = max(used_ids, default=0) + 1 

168 used_orders = [ 

169 value for value in (_positive_int(v) for v in frame["order_in_trial"]) if value 

170 ] 

171 next_order = max(used_orders, default=0) + 1 

172 for index in frame.index: 

173 if _positive_int(frame.at[index, "fixation_id"]) is None: 

174 frame.at[index, "fixation_id"] = next_id 

175 next_id += 1 

176 if _positive_int(frame.at[index, "order_in_trial"]) is None: 

177 frame.at[index, "order_in_trial"] = next_order 

178 next_order += 1 

179 return frame 

180 

181 

182def _is_scalar_cell(value: Any) -> bool: 

183 return value is None or ( 

184 isinstance(value, (str, int, float)) and not isinstance(value, bool) 

185 ) 

186 

187 

188def event_records_frame(records: Any) -> pd.DataFrame: 

189 """Build the normalized event table from untrusted JSON records. 

190 

191 ``records`` comes from a link or a file, so its shape is checked before 

192 pandas sees it: a list of objects, each naming at least one event column, 

193 with plain number/text/null values. Anything else raises ``ValueError`` 

194 with a reason, rather than a ``TypeError`` from deep inside pandas. 

195 """ 

196 if not isinstance(records, list): 

197 raise ValueError("Authored fixations must be a list of objects.") 

198 for number, record in enumerate(records, start=1): 

199 if not isinstance(record, dict): 

200 raise ValueError(f"Authored fixation {number} is not an object.") 

201 if not any(column in record for column in EVENT_COLUMNS): 

202 raise ValueError( 

203 f"Authored fixation {number} has none of: {', '.join(EVENT_COLUMNS)}." 

204 ) 

205 bad = sorted( 

206 str(key) 

207 for key in EVENT_COLUMNS 

208 if key in record and not _is_scalar_cell(record[key]) 

209 ) 

210 if bad: 

211 raise ValueError( 

212 f"Authored fixation {number} has a non-scalar {', '.join(bad)}." 

213 ) 

214 frame, _ = reconcile_event_table(pd.DataFrame(records).reset_index(drop=True)) 

215 return frame 

216 

217 

218def event_problems(words: pd.DataFrame, events: pd.DataFrame | None) -> list[str]: 

219 """Return actionable structural problems without silently renumbering rows.""" 

220 frame = normalize_event_table(events) 

221 problems: list[str] = [] 

222 ids = [_positive_int(value) for value in frame["fixation_id"]] 

223 orders = [_positive_int(value) for value in frame["order_in_trial"]] 

224 duplicate_ids = sorted({value for value in ids if value and ids.count(value) > 1}) 

225 duplicate_orders = sorted( 

226 {value for value in orders if value and orders.count(value) > 1} 

227 ) 

228 if duplicate_ids: 

229 problems.append( 

230 "Fixation id must be unique; duplicate: " 

231 + ", ".join(str(value) for value in duplicate_ids) 

232 + "." 

233 ) 

234 if duplicate_orders: 

235 problems.append( 

236 "Order must be unique; duplicate: " 

237 + ", ".join(str(value) for value in duplicate_orders) 

238 + "." 

239 ) 

240 unusable = unusable_event_rows(words, frame) 

241 if unusable: 

242 listed = ", ".join(str(number) for number in unusable[:10]) 

243 suffix = f" (+{len(unusable) - 10} more)" if len(unusable) > 10 else "" 

244 problems.append( 

245 f"Not drawn — no X/Y and no valid target word: row {listed}{suffix}." 

246 ) 

247 return problems 

248 

249 

250def _structural_problems(events: pd.DataFrame) -> list[str]: 

251 """Problems that make stable id/order reconciliation ambiguous.""" 

252 return [ 

253 problem 

254 for problem in event_problems(pd.DataFrame(), events) 

255 if problem.startswith(("Fixation id", "Order")) 

256 ] 

257 

258 

259def reconcile_event_table( 

260 events: pd.DataFrame | None, selected_fixation_id: int | None = None 

261) -> tuple[pd.DataFrame, int | None]: 

262 """Normalize table edits and preserve selection by stable fixation id.""" 

263 frame = normalize_event_table(events) 

264 problems = _structural_problems(frame) 

265 if problems: 

266 raise ValueError(" ".join(problems)) 

267 ids = {_positive_int(value) for value in frame["fixation_id"]} 

268 selected = selected_fixation_id if selected_fixation_id in ids else None 

269 return frame, selected 

270 

271 

272def apply_authoring_event( 

273 events: pd.DataFrame | None, 

274 event: Mapping[str, Any], 

275 *, 

276 selected_fixation_id: int | None = None, 

277) -> tuple[pd.DataFrame, int | None]: 

278 """Apply one compact canvas event keyed by stable ``fixation_id``.""" 

279 frame, selected = reconcile_event_table(events, selected_fixation_id) 

280 action = str(event.get("type", "")) 

281 fixation_id = _positive_int(event.get("fixation_id")) 

282 if action == "add": 

283 x, y = _number(event.get("x")), _number(event.get("y")) 

284 if x is None or y is None: 

285 raise ValueError("A new fixation needs finite X and Y coordinates.") 

286 existing_ids = [_positive_int(value) or 0 for value in frame["fixation_id"]] 

287 existing_orders = [ 

288 _positive_int(value) or 0 for value in frame["order_in_trial"] 

289 ] 

290 fixation_id = max(existing_ids, default=0) + 1 

291 frame.loc[len(frame)] = { 

292 "fixation_id": fixation_id, 

293 "order_in_trial": max(existing_orders, default=0) + 1, 

294 "word_id": None, 

295 "x": x, 

296 "y": y, 

297 "duration_ms": 220, 

298 } 

299 selected = fixation_id 

300 elif action in {"move", "update"}: 

301 if fixation_id is None: 

302 raise ValueError("The canvas edit has no valid fixation id.") 

303 matches = frame.index[ 

304 frame["fixation_id"].map(_positive_int).eq(fixation_id) 

305 ].tolist() 

306 if not matches: 

307 raise ValueError(f"Fixation {fixation_id} no longer exists.") 

308 allowed = ( 

309 {"x", "y"} 

310 if action == "move" 

311 else { 

312 "x", 

313 "y", 

314 "word_id", 

315 "duration_ms", 

316 "order_in_trial", 

317 } 

318 ) 

319 for field in allowed: 

320 if field in event: 

321 frame.at[matches[0], field] = event[field] 

322 selected = fixation_id 

323 elif action == "delete": 

324 if fixation_id is not None: 

325 frame = frame[ 

326 ~frame["fixation_id"].map(_positive_int).eq(fixation_id) 

327 ].reset_index(drop=True) 

328 selected = None if selected == fixation_id else selected 

329 elif action == "select": 

330 selected = fixation_id 

331 else: 

332 raise ValueError(f"Unknown authoring event: {action or '(blank)'}.") 

333 return reconcile_event_table(frame, selected) 

334 

335 

336def _word_key(text: Any) -> str: 

337 """A word as a target names it: letters and digits, case folded — so a 

338 punctuation or capitalisation fix leaves the target where it was.""" 

339 return "".join(char for char in str(text).casefold() if char.isalnum()) 

340 

341 

342def _target_keys(words: pd.DataFrame) -> dict[int, str]: 

343 if words.empty: 

344 return {} 

345 return { 

346 int(word_id): _word_key(text) 

347 for word_id, text in zip(words["word_id"], words["text"], strict=True) 

348 } 

349 

350 

351def stale_target_words( 

352 old_words: pd.DataFrame, new_words: pd.DataFrame, events: pd.DataFrame | None 

353) -> dict[int, tuple[int, str]]: 

354 """The fixations whose target word a text edit removed or replaced, as 

355 ``{fixation_id: (word_id, the word it named)}``. 

356 

357 Editing the stimulus keeps every authored fixation as it is — X/Y, order, 

358 duration and id. Only the optional target word can go out of date: its id 

359 may no longer exist, or now name a different word. Those are reported, not 

360 changed; a punctuation or capitalisation fix to the word is not a change. 

361 :func:`unresolved_targets` says which are still out of date later on. 

362 """ 

363 if events is None or events.empty: 

364 return {} 

365 before, after = _target_keys(old_words), _target_keys(new_words) 

366 stale: dict[int, tuple[int, str]] = {} 

367 for event in normalize_event_table(events).to_dict("records"): 

368 word_id = event_target_word(event) 

369 fixation_id = _positive_int(event.get("fixation_id")) 

370 if word_id is None or fixation_id is None or word_id not in before: 

371 continue 

372 if after.get(word_id) != before[word_id]: 

373 stale[fixation_id] = (word_id, before[word_id]) 

374 return stale 

375 

376 

377def unresolved_targets( 

378 stale: Mapping[int, tuple[int, str]], words: pd.DataFrame, events: pd.DataFrame 

379) -> dict[int, int]: 

380 """``{fixation_id: word_id}`` for the entries of :func:`stale_target_words` 

381 that are still out of date: the fixation exists, still targets that word id, 

382 and the text there is still not the word it named. Changing or clearing the 

383 target, deleting the fixation or undoing the text edit resolves one.""" 

384 if not stale or events is None or events.empty: 

385 return {} 

386 current = _target_keys(words) 

387 targets = { 

388 _positive_int(event.get("fixation_id")): event_target_word(event) 

389 for event in normalize_event_table(events).to_dict("records") 

390 } 

391 return { 

392 fixation_id: word_id 

393 for fixation_id, (word_id, named) in stale.items() 

394 if targets.get(fixation_id) == word_id and current.get(word_id) != named 

395 } 

396 

397 

398def destructive_change(before: pd.DataFrame | None, after: pd.DataFrame | None) -> bool: 

399 """Whether going from ``before`` to ``after`` lost authored work: a fixation 

400 removed, or one that is still there moved, retimed, reordered or retargeted. 

401 

402 Adding a fixation loses nothing, so it is not one. This is what decides when 

403 the authoring screen keeps the previous draft for **Restore previous draft**. 

404 """ 

405 old = normalize_event_table(before) 

406 new = normalize_event_table(after) 

407 if old.empty: 

408 return False 

409 

410 def keyed(frame: pd.DataFrame) -> dict[int, tuple]: 

411 rows = {} 

412 for event in frame.to_dict("records"): 

413 fixation_id = _positive_int(event.get("fixation_id")) 

414 rows[fixation_id] = ( 

415 _number(event.get("x")), 

416 _number(event.get("y")), 

417 _number(event.get("duration_ms")), 

418 _positive_int(event.get("order_in_trial")), 

419 event_target_word(event), 

420 ) 

421 return rows 

422 

423 old_rows, new_rows = keyed(old), keyed(new) 

424 return any(new_rows.get(key) != value for key, value in old_rows.items()) 

425 

426 

427def unusable_event_rows(words: pd.DataFrame, events: pd.DataFrame) -> list[int]: 

428 """Return rows that have neither usable X/Y nor a valid word fallback.""" 

429 if events is None or events.empty: 

430 return [] 

431 valid_words = ( 

432 set(words["word_id"].astype(int)) 

433 if not words.empty and "word_id" in words 

434 else set() 

435 ) 

436 unusable: list[int] = [] 

437 for number, event in enumerate(events.to_dict("records"), start=1): 

438 has_xy = ( 

439 _number(event.get("x")) is not None and _number(event.get("y")) is not None 

440 ) 

441 if not has_xy and event_target_word(event) not in valid_words: 

442 unusable.append(number) 

443 return unusable 

444 

445 

446def authored_fixations(words: pd.DataFrame, events: pd.DataFrame) -> pd.DataFrame: 

447 """Normalize authored events into the ordinary canonical fixation schema.""" 

448 columns = [ 

449 "participant_id", 

450 "trial_id", 

451 "text_id", 

452 "x", 

453 "y", 

454 "duration_ms", 

455 "timestamp_ms", 

456 "fixation_id", 

457 "word_id", 

458 "order_in_trial", 

459 ] 

460 if events is None or events.empty: 

461 return pd.DataFrame(columns=columns) 

462 frame, _ = reconcile_event_table(events) 

463 centers = pd.DataFrame(columns=["x", "y"]) 

464 if not words.empty: 

465 centers = words.set_index(words["word_id"].astype(int))[ 

466 ["x", "y", "width", "height"] 

467 ].copy() 

468 centers["x"] = centers["x"] + centers["width"] / 2 

469 centers["y"] = centers["y"] + centers["height"] / 2 

470 

471 frame = frame.assign( 

472 _order=frame["order_in_trial"].map(_positive_int), 

473 _row=range(len(frame)), 

474 ).sort_values(["_order", "_row"], kind="stable") 

475 rows: list[dict[str, Any]] = [] 

476 timestamp = 0.0 

477 for event in frame.to_dict("records"): 

478 word_id = event_target_word(event) 

479 x, y = _number(event.get("x")), _number(event.get("y")) 

480 if (x is None or y is None) and word_id in centers.index: 

481 x, y = float(centers.loc[word_id, "x"]), float(centers.loc[word_id, "y"]) 

482 if x is None or y is None: 

483 continue 

484 duration = _number(event.get("duration_ms")) 

485 duration = duration if duration is not None and duration > 0 else 220.0 

486 order = _positive_int(event.get("order_in_trial")) 

487 fixation_id = _positive_int(event.get("fixation_id")) 

488 rows.append( 

489 { 

490 "participant_id": "author", 

491 "trial_id": "authored-1", 

492 "text_id": "authored-text", 

493 "x": x, 

494 "y": y, 

495 "duration_ms": duration, 

496 "timestamp_ms": timestamp, 

497 "fixation_id": float(fixation_id), 

498 "word_id": float(word_id) if word_id in centers.index else float("nan"), 

499 "order_in_trial": int(order), 

500 } 

501 ) 

502 timestamp += duration 

503 return pd.DataFrame(rows, columns=columns) 

504 

505 

506def _json_records(events: pd.DataFrame) -> list[dict[str, Any]]: 

507 records: list[dict[str, Any]] = [] 

508 for row in normalize_event_table(events).to_dict("records"): 

509 records.append( 

510 { 

511 key: None 

512 if value is None or (isinstance(value, float) and math.isnan(value)) 

513 else value 

514 for key, value in row.items() 

515 } 

516 ) 

517 return records 

518 

519 

520def authoring_json( 

521 text: str, 

522 events: pd.DataFrame, 

523 *, 

524 layout: Mapping[str, int] | None = None, 

525) -> str: 

526 """Serialize a schema-2 source document for portable save/restore.""" 

527 layout_value = {**DEFAULT_LAYOUT, **dict(layout or {})} 

528 return json.dumps( 

529 { 

530 "schema": AUTHORING_SCHEMA, 

531 "text": text, 

532 "layout": layout_value, 

533 "fixations": _json_records(events), 

534 }, 

535 indent=2, 

536 ) 

537 

538 

539#: Schema 1 or 2 is accepted; the user is told only what the file is not. 

540_NOT_AUTHORING = "That file isn't an authoring file saved from this editor." 

541 

542 

543def parse_authoring_document(payload: str) -> AuthoringDocument: 

544 """Restore schema 2 or migrate a VIZ-20 schema-1 authoring document.""" 

545 try: 

546 value = json.loads(payload) 

547 except ValueError: 

548 value = None 

549 if not isinstance(value, dict): 

550 raise ValueError(_NOT_AUTHORING) 

551 schema = value.get("schema") 

552 if schema not in {1, AUTHORING_SCHEMA} or not isinstance(value.get("text"), str): 

553 raise ValueError(_NOT_AUTHORING) 

554 events = value.get("fixations", []) 

555 if not isinstance(events, list): 

556 raise ValueError("The authoring file's fixations must be a list.") 

557 frame = event_records_frame(events) 

558 layout = value.get("layout", {}) if schema == AUTHORING_SCHEMA else {} 

559 if not isinstance(layout, dict): 

560 raise ValueError("The authoring file's layout must be an object.") 

561 unknown = sorted(set(layout) - set(DEFAULT_LAYOUT)) 

562 if unknown: 

563 raise ValueError(f"Unknown authoring layout option: {', '.join(unknown)}.") 

564 try: 

565 normalized_layout = { 

566 key: int({**DEFAULT_LAYOUT, **layout}[key]) for key in DEFAULT_LAYOUT 

567 } 

568 except (TypeError, ValueError) as exc: 

569 raise ValueError("Authoring layout values must be whole numbers.") from exc 

570 return AuthoringDocument(value["text"], frame, normalized_layout)