spanmark 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
spanmark/_session.py ADDED
@@ -0,0 +1,896 @@
1
+ """Public annotation-session orchestration."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import warnings
7
+ from collections.abc import Iterator, Mapping, Sequence
8
+ from dataclasses import replace
9
+ from pathlib import Path
10
+ from types import TracebackType
11
+ from typing import Any, TypeAlias, cast
12
+
13
+ from filelock import FileLock, Timeout
14
+ from IPython.display import display
15
+ from typing_extensions import Self
16
+
17
+ from spanmark._autosave import AUTOSAVE_CHECKPOINT_BYTES, AutosaveOverlay
18
+ from spanmark._model import (
19
+ Document,
20
+ DocumentState,
21
+ WorkingSpan,
22
+ annotation_span_to_dict,
23
+ normalize_document,
24
+ suggestion_to_working_span,
25
+ validate_working_spans,
26
+ working_span_to_dict,
27
+ )
28
+ from spanmark._source import InMemoryDocumentSource, JsonlDocumentSource
29
+ from spanmark._storage import (
30
+ FileFingerprint,
31
+ atomic_write_jsonl_with_fingerprint,
32
+ fingerprint_file,
33
+ )
34
+ from spanmark._widget import SpanmarkWidget
35
+
36
+ _DocumentSource: TypeAlias = InMemoryDocumentSource | JsonlDocumentSource
37
+
38
+
39
+ class AnnotationSession:
40
+ """A one-way span annotation session.
41
+
42
+ spanmark keeps a complete annotated JSONL checkpoint at `output_path`.
43
+ Annotation changes are durably autosaved to a small hidden state overlay and
44
+ periodically checkpointed back into the JSONL. A clean close, explicit
45
+ `save()`, and workflow completion leave `output_path` fully current.
46
+
47
+ Every output record keeps the original input fields and gains a reserved
48
+ `annotation` object. To resume later, open that annotated JSONL file with
49
+ `from_jsonl()` and use the same path as `output_path`.
50
+
51
+ Args:
52
+ documents: Input documents to annotate. Each mapping must contain an explicit,
53
+ non-empty `id` and a `text` field, and document IDs must be unique.
54
+ Optional model pre-annotations belong under `suggestions`. Other input
55
+ fields are preserved in the output dataset.
56
+ labels: Non-empty sequence of unique span labels. Every suggestion and persisted
57
+ annotation span must use one of these labels.
58
+ output_path: JSONL annotated-dataset destination. The path must not already
59
+ exist for an in-memory session. spanmark creates it immediately, durably
60
+ autosaves annotation changes, and holds an exclusive lock for the path until
61
+ `close()` is called.
62
+ trim_whitespace: If true (default), remove leading and trailing whitespace
63
+ from newly selected text before creating a span. A selection containing only
64
+ whitespace is ignored.
65
+ allow_overlaps: If false (default), reject overlapping or nested spans and use
66
+ the classic inline-highlight UI. If true, allow overlaps/nesting and use the
67
+ colored annotation-rail UI.
68
+
69
+ Attributes:
70
+ labels: Validated annotation labels, stored as a tuple of strings.
71
+ output_path: Path to the annotated JSONL dataset.
72
+ allow_overlaps: Whether overlapping and nested spans are allowed.
73
+ widget: Annotation widget associated with the session.
74
+
75
+ Note:
76
+ During the main pass, Accept, Reject, and Ignore autosave and advance to the
77
+ next undecided document. Flag is an independent bookmark. Once every document
78
+ is decided, flagged documents can be reviewed in dataset order; clearing a
79
+ flag resolves that review item and advances to the next flagged document.
80
+ """
81
+
82
+ ANSWERS = frozenset({"accept", "reject", "ignore"})
83
+
84
+ def __init__(
85
+ self,
86
+ documents: Sequence[Mapping[str, Any]],
87
+ labels: Sequence[str],
88
+ output_path: os.PathLike[str] | Path,
89
+ *,
90
+ trim_whitespace: bool = True,
91
+ allow_overlaps: bool = False,
92
+ ) -> None:
93
+ source = InMemoryDocumentSource(documents)
94
+ self._initialize(
95
+ source=source,
96
+ labels=labels,
97
+ output_path=output_path,
98
+ trim_whitespace=trim_whitespace,
99
+ allow_overlaps=allow_overlaps,
100
+ source_is_output=False,
101
+ )
102
+
103
+ @classmethod
104
+ def from_jsonl(
105
+ cls,
106
+ path: os.PathLike[str] | Path,
107
+ labels: Sequence[str],
108
+ output_path: os.PathLike[str] | Path,
109
+ *,
110
+ trim_whitespace: bool = True,
111
+ allow_overlaps: bool = False,
112
+ ) -> Self:
113
+ """Create a session from a JSONL dataset.
114
+
115
+ For a new annotation run, `path` and `output_path` are different and
116
+ `output_path` must not already exist. spanmark immediately writes a
117
+ complete annotated copy there.
118
+
119
+ To resume, pass the previous annotated output as both `path` and
120
+ `output_path`. spanmark recovers any pending hidden autosave state,
121
+ checkpoints it into the JSONL, and starts at the first record whose
122
+ annotation decision is still unset. If every record is already decided,
123
+ it opens the completion state, where any remaining flags can be reviewed.
124
+
125
+ Args:
126
+ path: JSONL input dataset. Each non-empty line must contain an object with
127
+ an explicit, non-empty `id` and `text` field, and IDs must be
128
+ unique within the file.
129
+ labels: Non-empty sequence of unique span labels. Every suggestion and
130
+ persisted annotation span must use one of these labels.
131
+ output_path: JSONL annotated-dataset destination. For a new run this must be a
132
+ different, non-existing path. To resume, pass the same annotated
133
+ JSONL path for both `path` and `output_path`.
134
+ trim_whitespace: If true (default), remove leading and trailing whitespace
135
+ from newly selected text before creating a span. A selection containing
136
+ only whitespace is ignored.
137
+ allow_overlaps: If false (default), reject overlapping or nested spans and
138
+ use the classic inline-highlight UI. If true, allow overlaps/nesting
139
+ and use the colored annotation-rail UI.
140
+ """
141
+ input_path = Path(path)
142
+ output = Path(output_path)
143
+ source_is_output = input_path.resolve() == output.resolve()
144
+
145
+ if not source_is_output and output.exists():
146
+ raise FileExistsError(
147
+ f"{output} already exists. To resume it, pass that file as both "
148
+ "the JSONL input and output_path; otherwise choose a new output_path."
149
+ )
150
+
151
+ source = JsonlDocumentSource(input_path)
152
+ session = cls.__new__(cls)
153
+ session._initialize(
154
+ source=source,
155
+ labels=labels,
156
+ output_path=output,
157
+ trim_whitespace=trim_whitespace,
158
+ allow_overlaps=allow_overlaps,
159
+ source_is_output=source_is_output,
160
+ )
161
+ return session
162
+
163
+ def __enter__(self) -> Self:
164
+ self._ensure_open()
165
+ return self
166
+
167
+ def __exit__(
168
+ self,
169
+ exc_type: type[BaseException] | None,
170
+ exc_value: BaseException | None,
171
+ traceback: TracebackType | None,
172
+ ) -> None:
173
+ if exc_type is None:
174
+ self.close()
175
+ return
176
+
177
+ try:
178
+ self.close()
179
+ except Exception as close_error:
180
+ note = f"spanmark also failed to checkpoint while closing: {close_error}"
181
+ if hasattr(exc_value, "add_note"):
182
+ exc_value.add_note(note)
183
+ else:
184
+ warnings.warn(note, RuntimeWarning, stacklevel=2)
185
+
186
+ def close(self) -> None:
187
+ """Checkpoint pending autosaves and release the output-path lock."""
188
+ if self._closed:
189
+ return
190
+
191
+ try:
192
+ if self._autosave.exists:
193
+ self._checkpoint()
194
+ finally:
195
+ self._release_lock()
196
+
197
+ def display(self) -> None:
198
+ """Display the annotation UI once."""
199
+ self._ensure_open()
200
+ display(self.widget)
201
+
202
+ @property
203
+ def index(self) -> int:
204
+ """The index of the current document."""
205
+ return self._index
206
+
207
+ @property
208
+ def current_id(self) -> str:
209
+ """The identifier of the current document."""
210
+ return self._source.ids[self._index]
211
+
212
+ def decide(self, answer: str) -> None:
213
+ """Submit an Accept/Reject/Ignore decision for the current document."""
214
+ self._ensure_open()
215
+ answer = str(answer).lower()
216
+ if answer not in self.ANSWERS:
217
+ raise ValueError(
218
+ f"answer must be one of {sorted(self.ANSWERS)}, got {answer!r}"
219
+ )
220
+
221
+ doc_id = self.current_id
222
+ state = replace(
223
+ self._state[doc_id],
224
+ answer=answer,
225
+ materialized=True,
226
+ )
227
+ decision_history = [
228
+ saved_id for saved_id in self._decision_history if saved_id != doc_id
229
+ ]
230
+ decision_history.append(doc_id)
231
+
232
+ self._append_state(doc_id, state)
233
+ self._state[doc_id] = state
234
+ self._decision_history = decision_history
235
+
236
+ self._loading = True
237
+ try:
238
+ self.widget.answer = answer
239
+ self._sync_summary_traits()
240
+ finally:
241
+ self._loading = False
242
+
243
+ if self._reviewing_flagged:
244
+ status = "Decision updated — clear Flag when review is resolved."
245
+ self.widget.status = self._checkpoint_warning_if_large() or status
246
+ return
247
+
248
+ next_index = self._next_undecided_index(self._index)
249
+ if next_index is not None:
250
+ self._index = next_index
251
+ self._load_current()
252
+ warning = self._checkpoint_warning_if_large()
253
+ if warning is not None:
254
+ self.widget.status = warning
255
+ else:
256
+ self._sync_summary_traits()
257
+ self._checkpoint_with_status("All examples are complete.")
258
+
259
+ def undo_decision(self) -> None:
260
+ """Undo and reopen the immediately previous submitted example."""
261
+ self._ensure_open()
262
+ if self._reviewing_flagged:
263
+ self.widget.status = (
264
+ "Previous-decision Undo is unavailable during flagged review."
265
+ )
266
+ return
267
+ if not self._decision_history:
268
+ self.widget.status = "No previous submitted decision to undo."
269
+ return
270
+
271
+ doc_id = self._decision_history[-1]
272
+ state = replace(
273
+ self._state[doc_id],
274
+ answer="",
275
+ materialized=True,
276
+ )
277
+ decision_history = self._decision_history[:-1]
278
+
279
+ self._append_state(doc_id, state)
280
+ self._state[doc_id] = state
281
+ self._decision_history = decision_history
282
+ self._index = self._source.index_of(doc_id)
283
+
284
+ self._load_current()
285
+ status = "Previous decision undone — review this example again."
286
+ self.widget.status = self._checkpoint_warning_if_large() or status
287
+
288
+ def flag(self, value: bool = True) -> None:
289
+ """Set or clear the flag on the current document."""
290
+ self._ensure_open()
291
+ new_value = bool(value)
292
+ doc_id = self.current_id
293
+ current_state = self._state[doc_id]
294
+
295
+ if (
296
+ self._reviewing_flagged
297
+ and not new_value
298
+ and current_state.answer not in self.ANSWERS
299
+ ):
300
+ self._loading = True
301
+ try:
302
+ self.widget.flagged = True
303
+ finally:
304
+ self._loading = False
305
+ self.widget.status = (
306
+ "Choose Accept, Reject, or Ignore before clearing Flag."
307
+ )
308
+ return
309
+
310
+ state = replace(
311
+ current_state,
312
+ flagged=new_value,
313
+ materialized=True,
314
+ )
315
+
316
+ self._append_state(doc_id, state)
317
+ self._state[doc_id] = state
318
+
319
+ self._loading = True
320
+ try:
321
+ self.widget.flagged = new_value
322
+ self._sync_summary_traits()
323
+ finally:
324
+ self._loading = False
325
+
326
+ if self._reviewing_flagged and not new_value:
327
+ next_index = self._first_flagged_index()
328
+ if next_index is not None:
329
+ self._index = next_index
330
+ self._load_current()
331
+ status = (
332
+ "Review flagged examples — clear Flag when this item is resolved."
333
+ )
334
+ self.widget.status = self._checkpoint_warning_if_large() or status
335
+ else:
336
+ self._reviewing_flagged = False
337
+ self._sync_summary_traits()
338
+ self._checkpoint_with_status("Flagged review complete.")
339
+ return
340
+
341
+ warning = self._checkpoint_warning_if_large()
342
+ if warning is not None:
343
+ self.widget.status = warning
344
+
345
+ def records(self) -> list[dict[str, Any]]:
346
+ """Return the complete current annotated dataset."""
347
+ self._ensure_open()
348
+ return list(self._iter_records())
349
+
350
+ def save(self) -> Path:
351
+ """Checkpoint all current annotation state into `output_path`."""
352
+ self._ensure_open()
353
+ return self._checkpoint()
354
+
355
+ def summary(self) -> dict[str, Any]:
356
+ """Return compact annotation progress statistics."""
357
+ return {
358
+ "documents": len(self._source),
359
+ "started": sum(state.materialized for state in self._state.values()),
360
+ "decided": self._decided_count(),
361
+ "accept": self._answer_count("accept"),
362
+ "reject": self._answer_count("reject"),
363
+ "ignore": self._answer_count("ignore"),
364
+ "flagged": self._flagged_count(),
365
+ "spans": sum(len(state.spans) for state in self._state.values()),
366
+ "output_path": str(self.output_path),
367
+ }
368
+
369
+ def _initialize(
370
+ self,
371
+ *,
372
+ source: _DocumentSource,
373
+ labels: Sequence[str],
374
+ output_path: os.PathLike[str] | Path,
375
+ trim_whitespace: bool,
376
+ allow_overlaps: bool,
377
+ source_is_output: bool,
378
+ ) -> None:
379
+ if not labels:
380
+ raise ValueError("labels must contain at least one label")
381
+
382
+ self.labels = tuple(str(label) for label in labels)
383
+ if len(set(self.labels)) != len(self.labels):
384
+ raise ValueError("labels must be unique")
385
+
386
+ self.output_path = Path(output_path)
387
+ self.allow_overlaps = bool(allow_overlaps)
388
+ self._source = source
389
+ self._source_is_output = bool(source_is_output)
390
+ self._state: dict[str, DocumentState] = {}
391
+ self._loading = False
392
+ self._index = 0
393
+ self._closed = False
394
+ self._reviewing_flagged = False
395
+ self._autosave = AutosaveOverlay(self.output_path)
396
+ self._output_fingerprint: FileFingerprint | None = None
397
+ self._output_signature: tuple[int, int] | None = None
398
+
399
+ self.output_path.parent.mkdir(parents=True, exist_ok=True)
400
+ lock_path = self.output_path.with_name(f".{self.output_path.name}.lock")
401
+ self._output_lock = FileLock(lock_path)
402
+
403
+ try:
404
+ self._output_lock.acquire(timeout=0)
405
+ except Timeout as exc:
406
+ self._closed = True
407
+ raise RuntimeError(
408
+ f"{self.output_path} is already in use by another spanmark session"
409
+ ) from exc
410
+
411
+ try:
412
+ if not self._source_is_output and self.output_path.exists():
413
+ raise FileExistsError(
414
+ f"{self.output_path} already exists. To resume it, pass that "
415
+ "JSONL file as both input and output_path; otherwise choose a "
416
+ "new output_path."
417
+ )
418
+ if not self._source_is_output and self._autosave.exists:
419
+ raise RuntimeError(
420
+ f"Recovery autosave {self._autosave.path} exists without a "
421
+ f"resumable output at {self.output_path}. Move or remove the "
422
+ "orphaned autosave before starting a new session."
423
+ )
424
+
425
+ if self._source_is_output:
426
+ if not isinstance(self._source, JsonlDocumentSource):
427
+ raise RuntimeError("Only a JSONL source can be its own output")
428
+ self._source.refresh()
429
+ self._set_output_identity(fingerprint_file(self.output_path))
430
+
431
+ self._load_states()
432
+ if self._source_is_output and self._autosave.exists:
433
+ self._recover_autosave()
434
+
435
+ self._index = self._resume_index()
436
+ self._decision_history = self._resume_decision_history()
437
+
438
+ self.widget = SpanmarkWidget(
439
+ labels=list(self.labels),
440
+ trim_whitespace=trim_whitespace,
441
+ allow_overlaps=self.allow_overlaps,
442
+ )
443
+ self.widget.observe(self._on_spans, names="spans")
444
+ self.widget.observe(self._on_flagged, names="flagged")
445
+ self.widget.observe(self._on_event, names="event")
446
+
447
+ self._load_current()
448
+
449
+ # A new session starts with a complete JSONL checkpoint. On resume,
450
+ # this also folds any recovered autosave overlay back into the file.
451
+ self._checkpoint()
452
+ except BaseException:
453
+ self._release_lock()
454
+ raise
455
+
456
+ def _load_states(self) -> None:
457
+ for index, expected_id in enumerate(self._source.ids):
458
+ raw = self._source.record(index)
459
+ document = normalize_document(raw, index)
460
+ if document.id != expected_id:
461
+ raise RuntimeError(
462
+ f"Document source index expected {expected_id!r}, "
463
+ f"found {document.id!r}"
464
+ )
465
+ self._state[document.id] = self._state_from_record(document, raw)
466
+
467
+ def _state_from_record(
468
+ self,
469
+ document: Document,
470
+ raw: Mapping[str, Any],
471
+ ) -> DocumentState:
472
+ suggestion_spans = tuple(
473
+ suggestion_to_working_span(suggestion, index)
474
+ for index, suggestion in enumerate(document.suggestions)
475
+ )
476
+ suggestion_spans = validate_working_spans(
477
+ suggestion_spans,
478
+ text=document.text,
479
+ labels=self.labels,
480
+ allow_overlaps=self.allow_overlaps,
481
+ )
482
+
483
+ annotation = raw.get("annotation")
484
+ if annotation is None:
485
+ return DocumentState(spans=suggestion_spans)
486
+ if not isinstance(annotation, Mapping):
487
+ raise TypeError(f"Document {document.id!r} 'annotation' must be an object")
488
+
489
+ raw_answer = annotation.get("answer")
490
+ if raw_answer is None:
491
+ answer = ""
492
+ elif isinstance(raw_answer, str) and raw_answer in self.ANSWERS:
493
+ answer = raw_answer
494
+ else:
495
+ raise ValueError(
496
+ f"Document {document.id!r} annotation answer must be one of "
497
+ f"{sorted(self.ANSWERS)} or null"
498
+ )
499
+
500
+ raw_flagged = annotation.get("flagged", False)
501
+ if not isinstance(raw_flagged, bool):
502
+ raise TypeError(
503
+ f"Document {document.id!r} annotation flagged must be a boolean"
504
+ )
505
+
506
+ raw_spans = annotation.get("spans")
507
+ if raw_spans is None:
508
+ if answer or raw_flagged:
509
+ raise ValueError(
510
+ f"Document {document.id!r} annotation spans cannot be null "
511
+ "once a decision or flag has been recorded"
512
+ )
513
+ return DocumentState(spans=suggestion_spans)
514
+
515
+ if isinstance(raw_spans, (str, bytes, Mapping)) or not isinstance(
516
+ raw_spans, Sequence
517
+ ):
518
+ raise TypeError(
519
+ f"Document {document.id!r} annotation spans must be a list or null"
520
+ )
521
+
522
+ persisted_spans = cast(
523
+ Sequence[Mapping[str, Any] | WorkingSpan],
524
+ raw_spans,
525
+ )
526
+ spans = validate_working_spans(
527
+ persisted_spans,
528
+ text=document.text,
529
+ labels=self.labels,
530
+ allow_overlaps=self.allow_overlaps,
531
+ )
532
+ return DocumentState(
533
+ spans=spans,
534
+ answer=answer,
535
+ flagged=raw_flagged,
536
+ materialized=True,
537
+ )
538
+
539
+ def _recover_autosave(self) -> None:
540
+ recovery = self._autosave.recover()
541
+ if recovery is None:
542
+ return
543
+
544
+ current_fingerprint = self._output_fingerprint
545
+ if current_fingerprint is None:
546
+ raise RuntimeError("Cannot recover autosave without a JSONL checkpoint")
547
+
548
+ recovered_states: dict[str, DocumentState] = {}
549
+ for document_id, annotation in recovery.annotations.items():
550
+ if document_id not in self._state:
551
+ raise RuntimeError(
552
+ f"Autosave overlay {self._autosave.path} refers to unknown "
553
+ f"document {document_id!r}"
554
+ )
555
+ index = self._source.index_of(document_id)
556
+ document = self._source[index]
557
+ recovered_states[document_id] = self._state_from_record(
558
+ document,
559
+ {"annotation": annotation},
560
+ )
561
+
562
+ if recovery.base == current_fingerprint:
563
+ self._state.update(recovered_states)
564
+ return
565
+
566
+ if all(
567
+ self._state[document_id] == state
568
+ for document_id, state in recovered_states.items()
569
+ ):
570
+ # A checkpoint may have been atomically replaced before overlay cleanup.
571
+ # Discarding the stale overlay is safe only because every overlay entry
572
+ # is a complete per-document annotation state, not a patch: if every
573
+ # recovered state is already present in the JSONL, the overlay contains
574
+ # no additional information that could be lost. If overlay records ever
575
+ # become partial updates, this recovery rule must be redesigned.
576
+ return
577
+
578
+ raise RuntimeError(
579
+ f"Autosave overlay {self._autosave.path} was based on a different "
580
+ "JSONL checkpoint and cannot be safely recovered. The output may have "
581
+ "been modified outside spanmark."
582
+ )
583
+
584
+ def _current_doc(self) -> Document:
585
+ return self._source[self._index]
586
+
587
+ def _decided_count(self) -> int:
588
+ return sum(state.answer in self.ANSWERS for state in self._state.values())
589
+
590
+ def _answer_count(self, answer: str) -> int:
591
+ return sum(state.answer == answer for state in self._state.values())
592
+
593
+ def _flagged_count(self) -> int:
594
+ return sum(state.flagged for state in self._state.values())
595
+
596
+ def _is_complete(self) -> bool:
597
+ return self._decided_count() == len(self._source)
598
+
599
+ def _sync_summary_traits(self) -> None:
600
+ self.widget.decided_count = self._decided_count()
601
+ self.widget.accept_count = self._answer_count("accept")
602
+ self.widget.reject_count = self._answer_count("reject")
603
+ self.widget.ignore_count = self._answer_count("ignore")
604
+ self.widget.flagged_count = self._flagged_count()
605
+ self.widget.complete = self._is_complete()
606
+ self.widget.reviewing_flagged = self._reviewing_flagged
607
+ self.widget.can_undo = (
608
+ bool(self._decision_history) and not self._reviewing_flagged
609
+ )
610
+
611
+ def _resume_index(self) -> int:
612
+ for index, document_id in enumerate(self._source.ids):
613
+ if self._state[document_id].answer not in self.ANSWERS:
614
+ return index
615
+ return len(self._source) - 1
616
+
617
+ def _resume_decision_history(self) -> list[str]:
618
+ current_id = self._source.ids[self._index]
619
+ current_decided = self._state[current_id].answer in self.ANSWERS
620
+ stop = self._index + 1 if current_decided else self._index
621
+
622
+ # JSONL stores decisions but not action chronology. Reconstructing history
623
+ # in dataset order matches the normal one-way pass, but cannot reproduce
624
+ # arbitrary pre-close edit/redecision order exactly.
625
+ return [
626
+ document_id
627
+ for document_id in self._source.ids[:stop]
628
+ if self._state[document_id].answer in self.ANSWERS
629
+ ]
630
+
631
+ def _next_undecided_index(self, after: int) -> int | None:
632
+ for index in range(after + 1, len(self._source)):
633
+ document_id = self._source.ids[index]
634
+ if self._state[document_id].answer not in self.ANSWERS:
635
+ return index
636
+ return None
637
+
638
+ def _first_flagged_index(self) -> int | None:
639
+ for index, document_id in enumerate(self._source.ids):
640
+ if self._state[document_id].flagged:
641
+ return index
642
+ return None
643
+
644
+ def _start_flagged_review(self) -> None:
645
+ if not self._is_complete():
646
+ self.widget.status = (
647
+ "Finish the main annotation pass before reviewing flags."
648
+ )
649
+ return
650
+
651
+ first_index = self._first_flagged_index()
652
+ if first_index is None:
653
+ self.widget.status = "There are no flagged examples to review."
654
+ return
655
+
656
+ self._reviewing_flagged = True
657
+ self._index = first_index
658
+ self._load_current()
659
+ self.widget.status = (
660
+ "Review flagged examples — clear Flag when this item is resolved."
661
+ )
662
+
663
+ def _load_current(self) -> None:
664
+ document = self._current_doc()
665
+ self._set_widget_document(document, self._state[document.id])
666
+
667
+ def _set_widget_document(
668
+ self,
669
+ document: Document,
670
+ state: DocumentState,
671
+ *,
672
+ status: str = "",
673
+ ) -> None:
674
+ self._loading = True
675
+ try:
676
+ self.widget.doc_id = document.id
677
+ self.widget.text = document.text
678
+ self.widget.index = self._index
679
+ self.widget.total = len(self._source)
680
+ self._set_widget_annotation_state(state, status=status)
681
+ finally:
682
+ self._loading = False
683
+
684
+ def _set_widget_annotation_state(
685
+ self,
686
+ state: DocumentState,
687
+ *,
688
+ status: str = "",
689
+ ) -> None:
690
+ self.widget.spans = [working_span_to_dict(span) for span in state.spans]
691
+ self.widget.answer = state.answer
692
+ self.widget.flagged = state.flagged
693
+ self._sync_summary_traits()
694
+ self.widget.status = status
695
+
696
+ def _restore_current_annotation_state(self, *, status: str) -> None:
697
+ self._loading = True
698
+ try:
699
+ self._set_widget_annotation_state(
700
+ self._state[self.current_id],
701
+ status=status,
702
+ )
703
+ finally:
704
+ self._loading = False
705
+
706
+ def _on_spans(self, change: Mapping[str, Any]) -> None:
707
+ if self._loading or self._closed:
708
+ return
709
+
710
+ doc_id = self.current_id
711
+ previous_state = self._state[doc_id]
712
+
713
+ try:
714
+ spans = validate_working_spans(
715
+ change["new"],
716
+ text=self.widget.text,
717
+ labels=self.labels,
718
+ allow_overlaps=self.allow_overlaps,
719
+ )
720
+ except Exception as exc:
721
+ self._restore_current_annotation_state(
722
+ status=f"Span edit rejected: {exc}",
723
+ )
724
+ return
725
+
726
+ state = replace(
727
+ previous_state,
728
+ spans=spans,
729
+ materialized=True,
730
+ )
731
+ decision_history = self._decision_history
732
+ status = ""
733
+
734
+ # A decision certifies the exact current span set. Any edit invalidates
735
+ # the previous decision while preserving Flag.
736
+ if state.answer in self.ANSWERS:
737
+ state = replace(state, answer="")
738
+ decision_history = [
739
+ saved_id for saved_id in self._decision_history if saved_id != doc_id
740
+ ]
741
+ if self._reviewing_flagged:
742
+ status = "Span edited — choose Accept, Reject, or Ignore before clearing Flag."
743
+ else:
744
+ status = "Span edited — previous decision cleared."
745
+
746
+ try:
747
+ self._append_state(doc_id, state)
748
+ except Exception as exc:
749
+ self._restore_current_annotation_state(
750
+ status=f"Could not save span edit: {exc}",
751
+ )
752
+ return
753
+
754
+ self._state[doc_id] = state
755
+ self._decision_history = decision_history
756
+ self._loading = True
757
+ try:
758
+ self._set_widget_annotation_state(state, status=status)
759
+ finally:
760
+ self._loading = False
761
+
762
+ warning = self._checkpoint_warning_if_large()
763
+ if warning is not None:
764
+ self.widget.status = warning
765
+
766
+ def _on_flagged(self, change: Mapping[str, Any]) -> None:
767
+ if self._loading or self._closed:
768
+ return
769
+
770
+ previous_state = self._state[self.current_id]
771
+ try:
772
+ self.flag(bool(change["new"]))
773
+ except Exception as exc:
774
+ self._loading = True
775
+ try:
776
+ self.widget.flagged = previous_state.flagged
777
+ self._sync_summary_traits()
778
+ self.widget.status = f"Could not save Flag change: {exc}"
779
+ finally:
780
+ self._loading = False
781
+
782
+ def _on_event(self, change: Mapping[str, Any]) -> None:
783
+ if self._loading or self._closed:
784
+ return
785
+
786
+ event = change["new"] or {}
787
+ kind = event.get("type")
788
+
789
+ try:
790
+ if kind == "decision":
791
+ self.decide(str(event.get("answer") or ""))
792
+ elif kind == "undo_decision":
793
+ self.undo_decision()
794
+ elif kind == "review_flagged":
795
+ self._start_flagged_review()
796
+ except Exception as exc:
797
+ self.widget.status = f"Could not apply action: {exc}"
798
+
799
+ def _append_state(self, document_id: str, state: DocumentState) -> None:
800
+ self._check_source_unchanged()
801
+ self._check_output_unchanged()
802
+
803
+ base = self._output_fingerprint
804
+ if base is None:
805
+ raise RuntimeError("No JSONL checkpoint is available for autosave")
806
+
807
+ self._autosave.append(
808
+ base=base,
809
+ document_id=document_id,
810
+ annotation=self._annotation_for_state(state),
811
+ )
812
+
813
+ def _checkpoint(self) -> Path:
814
+ self._check_source_unchanged()
815
+ self._check_output_unchanged()
816
+
817
+ result, fingerprint = atomic_write_jsonl_with_fingerprint(
818
+ self.output_path,
819
+ self._iter_records(),
820
+ )
821
+
822
+ if self._source_is_output:
823
+ if not isinstance(self._source, JsonlDocumentSource):
824
+ raise RuntimeError("Only a JSONL source can be its own output")
825
+ self._source.refresh()
826
+
827
+ self._set_output_identity(fingerprint)
828
+ self._autosave.discard()
829
+ return result
830
+
831
+ def _checkpoint_warning_if_large(self) -> str | None:
832
+ if self._autosave.size < AUTOSAVE_CHECKPOINT_BYTES:
833
+ return None
834
+
835
+ try:
836
+ self._checkpoint()
837
+ except Exception as exc:
838
+ return f"Autosaved, but JSONL checkpoint failed: {exc}"
839
+ return None
840
+
841
+ def _checkpoint_with_status(self, status: str) -> None:
842
+ try:
843
+ self._checkpoint()
844
+ except Exception as exc:
845
+ self.widget.status = (
846
+ f"{status} Autosaved, but JSONL checkpoint failed: {exc}"
847
+ )
848
+ else:
849
+ self.widget.status = status
850
+
851
+ def _check_source_unchanged(self) -> None:
852
+ if isinstance(self._source, JsonlDocumentSource):
853
+ self._source.check_unchanged()
854
+
855
+ def _check_output_unchanged(self) -> None:
856
+ if self._output_signature is None:
857
+ return
858
+
859
+ stat = self.output_path.stat()
860
+ current = stat.st_size, stat.st_mtime_ns
861
+ if current != self._output_signature:
862
+ raise RuntimeError(
863
+ f"JSONL output {self.output_path} changed while the session was open"
864
+ )
865
+
866
+ def _set_output_identity(self, fingerprint: FileFingerprint) -> None:
867
+ stat = self.output_path.stat()
868
+ self._output_fingerprint = fingerprint
869
+ self._output_signature = stat.st_size, stat.st_mtime_ns
870
+
871
+ def _annotation_for_state(self, state: DocumentState) -> dict[str, Any]:
872
+ return {
873
+ "spans": (
874
+ [annotation_span_to_dict(span) for span in state.spans]
875
+ if state.materialized
876
+ else None
877
+ ),
878
+ "answer": state.answer or None,
879
+ "flagged": state.flagged,
880
+ }
881
+
882
+ def _iter_records(self) -> Iterator[dict[str, Any]]:
883
+ for index, document_id in enumerate(self._source.ids):
884
+ record = dict(self._source.record(index))
885
+ record["annotation"] = self._annotation_for_state(self._state[document_id])
886
+ yield record
887
+
888
+ def _release_lock(self) -> None:
889
+ if self._closed:
890
+ return
891
+ self._output_lock.release()
892
+ self._closed = True
893
+
894
+ def _ensure_open(self) -> None:
895
+ if self._closed:
896
+ raise RuntimeError("This spanmark session is closed")