spanmark 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spanmark/__init__.py +17 -0
- spanmark/_autosave.py +237 -0
- spanmark/_model.py +290 -0
- spanmark/_session.py +896 -0
- spanmark/_source.py +179 -0
- spanmark/_storage.py +97 -0
- spanmark/_version.py +24 -0
- spanmark/_widget.py +41 -0
- spanmark/static/widget.css +308 -0
- spanmark/static/widget.js +851 -0
- spanmark-0.1.0.dist-info/METADATA +436 -0
- spanmark-0.1.0.dist-info/RECORD +14 -0
- spanmark-0.1.0.dist-info/WHEEL +4 -0
- spanmark-0.1.0.dist-info/licenses/LICENSE +21 -0
spanmark/_session.py
ADDED
|
@@ -0,0 +1,896 @@
|
|
|
1
|
+
"""Public annotation-session orchestration."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import warnings
|
|
7
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
8
|
+
from dataclasses import replace
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from types import TracebackType
|
|
11
|
+
from typing import Any, TypeAlias, cast
|
|
12
|
+
|
|
13
|
+
from filelock import FileLock, Timeout
|
|
14
|
+
from IPython.display import display
|
|
15
|
+
from typing_extensions import Self
|
|
16
|
+
|
|
17
|
+
from spanmark._autosave import AUTOSAVE_CHECKPOINT_BYTES, AutosaveOverlay
|
|
18
|
+
from spanmark._model import (
|
|
19
|
+
Document,
|
|
20
|
+
DocumentState,
|
|
21
|
+
WorkingSpan,
|
|
22
|
+
annotation_span_to_dict,
|
|
23
|
+
normalize_document,
|
|
24
|
+
suggestion_to_working_span,
|
|
25
|
+
validate_working_spans,
|
|
26
|
+
working_span_to_dict,
|
|
27
|
+
)
|
|
28
|
+
from spanmark._source import InMemoryDocumentSource, JsonlDocumentSource
|
|
29
|
+
from spanmark._storage import (
|
|
30
|
+
FileFingerprint,
|
|
31
|
+
atomic_write_jsonl_with_fingerprint,
|
|
32
|
+
fingerprint_file,
|
|
33
|
+
)
|
|
34
|
+
from spanmark._widget import SpanmarkWidget
|
|
35
|
+
|
|
36
|
+
_DocumentSource: TypeAlias = InMemoryDocumentSource | JsonlDocumentSource
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class AnnotationSession:
|
|
40
|
+
"""A one-way span annotation session.
|
|
41
|
+
|
|
42
|
+
spanmark keeps a complete annotated JSONL checkpoint at `output_path`.
|
|
43
|
+
Annotation changes are durably autosaved to a small hidden state overlay and
|
|
44
|
+
periodically checkpointed back into the JSONL. A clean close, explicit
|
|
45
|
+
`save()`, and workflow completion leave `output_path` fully current.
|
|
46
|
+
|
|
47
|
+
Every output record keeps the original input fields and gains a reserved
|
|
48
|
+
`annotation` object. To resume later, open that annotated JSONL file with
|
|
49
|
+
`from_jsonl()` and use the same path as `output_path`.
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
documents: Input documents to annotate. Each mapping must contain an explicit,
|
|
53
|
+
non-empty `id` and a `text` field, and document IDs must be unique.
|
|
54
|
+
Optional model pre-annotations belong under `suggestions`. Other input
|
|
55
|
+
fields are preserved in the output dataset.
|
|
56
|
+
labels: Non-empty sequence of unique span labels. Every suggestion and persisted
|
|
57
|
+
annotation span must use one of these labels.
|
|
58
|
+
output_path: JSONL annotated-dataset destination. The path must not already
|
|
59
|
+
exist for an in-memory session. spanmark creates it immediately, durably
|
|
60
|
+
autosaves annotation changes, and holds an exclusive lock for the path until
|
|
61
|
+
`close()` is called.
|
|
62
|
+
trim_whitespace: If true (default), remove leading and trailing whitespace
|
|
63
|
+
from newly selected text before creating a span. A selection containing only
|
|
64
|
+
whitespace is ignored.
|
|
65
|
+
allow_overlaps: If false (default), reject overlapping or nested spans and use
|
|
66
|
+
the classic inline-highlight UI. If true, allow overlaps/nesting and use the
|
|
67
|
+
colored annotation-rail UI.
|
|
68
|
+
|
|
69
|
+
Attributes:
|
|
70
|
+
labels: Validated annotation labels, stored as a tuple of strings.
|
|
71
|
+
output_path: Path to the annotated JSONL dataset.
|
|
72
|
+
allow_overlaps: Whether overlapping and nested spans are allowed.
|
|
73
|
+
widget: Annotation widget associated with the session.
|
|
74
|
+
|
|
75
|
+
Note:
|
|
76
|
+
During the main pass, Accept, Reject, and Ignore autosave and advance to the
|
|
77
|
+
next undecided document. Flag is an independent bookmark. Once every document
|
|
78
|
+
is decided, flagged documents can be reviewed in dataset order; clearing a
|
|
79
|
+
flag resolves that review item and advances to the next flagged document.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
ANSWERS = frozenset({"accept", "reject", "ignore"})
|
|
83
|
+
|
|
84
|
+
def __init__(
|
|
85
|
+
self,
|
|
86
|
+
documents: Sequence[Mapping[str, Any]],
|
|
87
|
+
labels: Sequence[str],
|
|
88
|
+
output_path: os.PathLike[str] | Path,
|
|
89
|
+
*,
|
|
90
|
+
trim_whitespace: bool = True,
|
|
91
|
+
allow_overlaps: bool = False,
|
|
92
|
+
) -> None:
|
|
93
|
+
source = InMemoryDocumentSource(documents)
|
|
94
|
+
self._initialize(
|
|
95
|
+
source=source,
|
|
96
|
+
labels=labels,
|
|
97
|
+
output_path=output_path,
|
|
98
|
+
trim_whitespace=trim_whitespace,
|
|
99
|
+
allow_overlaps=allow_overlaps,
|
|
100
|
+
source_is_output=False,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
@classmethod
|
|
104
|
+
def from_jsonl(
|
|
105
|
+
cls,
|
|
106
|
+
path: os.PathLike[str] | Path,
|
|
107
|
+
labels: Sequence[str],
|
|
108
|
+
output_path: os.PathLike[str] | Path,
|
|
109
|
+
*,
|
|
110
|
+
trim_whitespace: bool = True,
|
|
111
|
+
allow_overlaps: bool = False,
|
|
112
|
+
) -> Self:
|
|
113
|
+
"""Create a session from a JSONL dataset.
|
|
114
|
+
|
|
115
|
+
For a new annotation run, `path` and `output_path` are different and
|
|
116
|
+
`output_path` must not already exist. spanmark immediately writes a
|
|
117
|
+
complete annotated copy there.
|
|
118
|
+
|
|
119
|
+
To resume, pass the previous annotated output as both `path` and
|
|
120
|
+
`output_path`. spanmark recovers any pending hidden autosave state,
|
|
121
|
+
checkpoints it into the JSONL, and starts at the first record whose
|
|
122
|
+
annotation decision is still unset. If every record is already decided,
|
|
123
|
+
it opens the completion state, where any remaining flags can be reviewed.
|
|
124
|
+
|
|
125
|
+
Args:
|
|
126
|
+
path: JSONL input dataset. Each non-empty line must contain an object with
|
|
127
|
+
an explicit, non-empty `id` and `text` field, and IDs must be
|
|
128
|
+
unique within the file.
|
|
129
|
+
labels: Non-empty sequence of unique span labels. Every suggestion and
|
|
130
|
+
persisted annotation span must use one of these labels.
|
|
131
|
+
output_path: JSONL annotated-dataset destination. For a new run this must be a
|
|
132
|
+
different, non-existing path. To resume, pass the same annotated
|
|
133
|
+
JSONL path for both `path` and `output_path`.
|
|
134
|
+
trim_whitespace: If true (default), remove leading and trailing whitespace
|
|
135
|
+
from newly selected text before creating a span. A selection containing
|
|
136
|
+
only whitespace is ignored.
|
|
137
|
+
allow_overlaps: If false (default), reject overlapping or nested spans and
|
|
138
|
+
use the classic inline-highlight UI. If true, allow overlaps/nesting
|
|
139
|
+
and use the colored annotation-rail UI.
|
|
140
|
+
"""
|
|
141
|
+
input_path = Path(path)
|
|
142
|
+
output = Path(output_path)
|
|
143
|
+
source_is_output = input_path.resolve() == output.resolve()
|
|
144
|
+
|
|
145
|
+
if not source_is_output and output.exists():
|
|
146
|
+
raise FileExistsError(
|
|
147
|
+
f"{output} already exists. To resume it, pass that file as both "
|
|
148
|
+
"the JSONL input and output_path; otherwise choose a new output_path."
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
source = JsonlDocumentSource(input_path)
|
|
152
|
+
session = cls.__new__(cls)
|
|
153
|
+
session._initialize(
|
|
154
|
+
source=source,
|
|
155
|
+
labels=labels,
|
|
156
|
+
output_path=output,
|
|
157
|
+
trim_whitespace=trim_whitespace,
|
|
158
|
+
allow_overlaps=allow_overlaps,
|
|
159
|
+
source_is_output=source_is_output,
|
|
160
|
+
)
|
|
161
|
+
return session
|
|
162
|
+
|
|
163
|
+
def __enter__(self) -> Self:
|
|
164
|
+
self._ensure_open()
|
|
165
|
+
return self
|
|
166
|
+
|
|
167
|
+
def __exit__(
|
|
168
|
+
self,
|
|
169
|
+
exc_type: type[BaseException] | None,
|
|
170
|
+
exc_value: BaseException | None,
|
|
171
|
+
traceback: TracebackType | None,
|
|
172
|
+
) -> None:
|
|
173
|
+
if exc_type is None:
|
|
174
|
+
self.close()
|
|
175
|
+
return
|
|
176
|
+
|
|
177
|
+
try:
|
|
178
|
+
self.close()
|
|
179
|
+
except Exception as close_error:
|
|
180
|
+
note = f"spanmark also failed to checkpoint while closing: {close_error}"
|
|
181
|
+
if hasattr(exc_value, "add_note"):
|
|
182
|
+
exc_value.add_note(note)
|
|
183
|
+
else:
|
|
184
|
+
warnings.warn(note, RuntimeWarning, stacklevel=2)
|
|
185
|
+
|
|
186
|
+
def close(self) -> None:
|
|
187
|
+
"""Checkpoint pending autosaves and release the output-path lock."""
|
|
188
|
+
if self._closed:
|
|
189
|
+
return
|
|
190
|
+
|
|
191
|
+
try:
|
|
192
|
+
if self._autosave.exists:
|
|
193
|
+
self._checkpoint()
|
|
194
|
+
finally:
|
|
195
|
+
self._release_lock()
|
|
196
|
+
|
|
197
|
+
def display(self) -> None:
|
|
198
|
+
"""Display the annotation UI once."""
|
|
199
|
+
self._ensure_open()
|
|
200
|
+
display(self.widget)
|
|
201
|
+
|
|
202
|
+
@property
|
|
203
|
+
def index(self) -> int:
|
|
204
|
+
"""The index of the current document."""
|
|
205
|
+
return self._index
|
|
206
|
+
|
|
207
|
+
@property
|
|
208
|
+
def current_id(self) -> str:
|
|
209
|
+
"""The identifier of the current document."""
|
|
210
|
+
return self._source.ids[self._index]
|
|
211
|
+
|
|
212
|
+
def decide(self, answer: str) -> None:
|
|
213
|
+
"""Submit an Accept/Reject/Ignore decision for the current document."""
|
|
214
|
+
self._ensure_open()
|
|
215
|
+
answer = str(answer).lower()
|
|
216
|
+
if answer not in self.ANSWERS:
|
|
217
|
+
raise ValueError(
|
|
218
|
+
f"answer must be one of {sorted(self.ANSWERS)}, got {answer!r}"
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
doc_id = self.current_id
|
|
222
|
+
state = replace(
|
|
223
|
+
self._state[doc_id],
|
|
224
|
+
answer=answer,
|
|
225
|
+
materialized=True,
|
|
226
|
+
)
|
|
227
|
+
decision_history = [
|
|
228
|
+
saved_id for saved_id in self._decision_history if saved_id != doc_id
|
|
229
|
+
]
|
|
230
|
+
decision_history.append(doc_id)
|
|
231
|
+
|
|
232
|
+
self._append_state(doc_id, state)
|
|
233
|
+
self._state[doc_id] = state
|
|
234
|
+
self._decision_history = decision_history
|
|
235
|
+
|
|
236
|
+
self._loading = True
|
|
237
|
+
try:
|
|
238
|
+
self.widget.answer = answer
|
|
239
|
+
self._sync_summary_traits()
|
|
240
|
+
finally:
|
|
241
|
+
self._loading = False
|
|
242
|
+
|
|
243
|
+
if self._reviewing_flagged:
|
|
244
|
+
status = "Decision updated — clear Flag when review is resolved."
|
|
245
|
+
self.widget.status = self._checkpoint_warning_if_large() or status
|
|
246
|
+
return
|
|
247
|
+
|
|
248
|
+
next_index = self._next_undecided_index(self._index)
|
|
249
|
+
if next_index is not None:
|
|
250
|
+
self._index = next_index
|
|
251
|
+
self._load_current()
|
|
252
|
+
warning = self._checkpoint_warning_if_large()
|
|
253
|
+
if warning is not None:
|
|
254
|
+
self.widget.status = warning
|
|
255
|
+
else:
|
|
256
|
+
self._sync_summary_traits()
|
|
257
|
+
self._checkpoint_with_status("All examples are complete.")
|
|
258
|
+
|
|
259
|
+
def undo_decision(self) -> None:
|
|
260
|
+
"""Undo and reopen the immediately previous submitted example."""
|
|
261
|
+
self._ensure_open()
|
|
262
|
+
if self._reviewing_flagged:
|
|
263
|
+
self.widget.status = (
|
|
264
|
+
"Previous-decision Undo is unavailable during flagged review."
|
|
265
|
+
)
|
|
266
|
+
return
|
|
267
|
+
if not self._decision_history:
|
|
268
|
+
self.widget.status = "No previous submitted decision to undo."
|
|
269
|
+
return
|
|
270
|
+
|
|
271
|
+
doc_id = self._decision_history[-1]
|
|
272
|
+
state = replace(
|
|
273
|
+
self._state[doc_id],
|
|
274
|
+
answer="",
|
|
275
|
+
materialized=True,
|
|
276
|
+
)
|
|
277
|
+
decision_history = self._decision_history[:-1]
|
|
278
|
+
|
|
279
|
+
self._append_state(doc_id, state)
|
|
280
|
+
self._state[doc_id] = state
|
|
281
|
+
self._decision_history = decision_history
|
|
282
|
+
self._index = self._source.index_of(doc_id)
|
|
283
|
+
|
|
284
|
+
self._load_current()
|
|
285
|
+
status = "Previous decision undone — review this example again."
|
|
286
|
+
self.widget.status = self._checkpoint_warning_if_large() or status
|
|
287
|
+
|
|
288
|
+
def flag(self, value: bool = True) -> None:
|
|
289
|
+
"""Set or clear the flag on the current document."""
|
|
290
|
+
self._ensure_open()
|
|
291
|
+
new_value = bool(value)
|
|
292
|
+
doc_id = self.current_id
|
|
293
|
+
current_state = self._state[doc_id]
|
|
294
|
+
|
|
295
|
+
if (
|
|
296
|
+
self._reviewing_flagged
|
|
297
|
+
and not new_value
|
|
298
|
+
and current_state.answer not in self.ANSWERS
|
|
299
|
+
):
|
|
300
|
+
self._loading = True
|
|
301
|
+
try:
|
|
302
|
+
self.widget.flagged = True
|
|
303
|
+
finally:
|
|
304
|
+
self._loading = False
|
|
305
|
+
self.widget.status = (
|
|
306
|
+
"Choose Accept, Reject, or Ignore before clearing Flag."
|
|
307
|
+
)
|
|
308
|
+
return
|
|
309
|
+
|
|
310
|
+
state = replace(
|
|
311
|
+
current_state,
|
|
312
|
+
flagged=new_value,
|
|
313
|
+
materialized=True,
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
self._append_state(doc_id, state)
|
|
317
|
+
self._state[doc_id] = state
|
|
318
|
+
|
|
319
|
+
self._loading = True
|
|
320
|
+
try:
|
|
321
|
+
self.widget.flagged = new_value
|
|
322
|
+
self._sync_summary_traits()
|
|
323
|
+
finally:
|
|
324
|
+
self._loading = False
|
|
325
|
+
|
|
326
|
+
if self._reviewing_flagged and not new_value:
|
|
327
|
+
next_index = self._first_flagged_index()
|
|
328
|
+
if next_index is not None:
|
|
329
|
+
self._index = next_index
|
|
330
|
+
self._load_current()
|
|
331
|
+
status = (
|
|
332
|
+
"Review flagged examples — clear Flag when this item is resolved."
|
|
333
|
+
)
|
|
334
|
+
self.widget.status = self._checkpoint_warning_if_large() or status
|
|
335
|
+
else:
|
|
336
|
+
self._reviewing_flagged = False
|
|
337
|
+
self._sync_summary_traits()
|
|
338
|
+
self._checkpoint_with_status("Flagged review complete.")
|
|
339
|
+
return
|
|
340
|
+
|
|
341
|
+
warning = self._checkpoint_warning_if_large()
|
|
342
|
+
if warning is not None:
|
|
343
|
+
self.widget.status = warning
|
|
344
|
+
|
|
345
|
+
def records(self) -> list[dict[str, Any]]:
|
|
346
|
+
"""Return the complete current annotated dataset."""
|
|
347
|
+
self._ensure_open()
|
|
348
|
+
return list(self._iter_records())
|
|
349
|
+
|
|
350
|
+
def save(self) -> Path:
|
|
351
|
+
"""Checkpoint all current annotation state into `output_path`."""
|
|
352
|
+
self._ensure_open()
|
|
353
|
+
return self._checkpoint()
|
|
354
|
+
|
|
355
|
+
def summary(self) -> dict[str, Any]:
|
|
356
|
+
"""Return compact annotation progress statistics."""
|
|
357
|
+
return {
|
|
358
|
+
"documents": len(self._source),
|
|
359
|
+
"started": sum(state.materialized for state in self._state.values()),
|
|
360
|
+
"decided": self._decided_count(),
|
|
361
|
+
"accept": self._answer_count("accept"),
|
|
362
|
+
"reject": self._answer_count("reject"),
|
|
363
|
+
"ignore": self._answer_count("ignore"),
|
|
364
|
+
"flagged": self._flagged_count(),
|
|
365
|
+
"spans": sum(len(state.spans) for state in self._state.values()),
|
|
366
|
+
"output_path": str(self.output_path),
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
def _initialize(
|
|
370
|
+
self,
|
|
371
|
+
*,
|
|
372
|
+
source: _DocumentSource,
|
|
373
|
+
labels: Sequence[str],
|
|
374
|
+
output_path: os.PathLike[str] | Path,
|
|
375
|
+
trim_whitespace: bool,
|
|
376
|
+
allow_overlaps: bool,
|
|
377
|
+
source_is_output: bool,
|
|
378
|
+
) -> None:
|
|
379
|
+
if not labels:
|
|
380
|
+
raise ValueError("labels must contain at least one label")
|
|
381
|
+
|
|
382
|
+
self.labels = tuple(str(label) for label in labels)
|
|
383
|
+
if len(set(self.labels)) != len(self.labels):
|
|
384
|
+
raise ValueError("labels must be unique")
|
|
385
|
+
|
|
386
|
+
self.output_path = Path(output_path)
|
|
387
|
+
self.allow_overlaps = bool(allow_overlaps)
|
|
388
|
+
self._source = source
|
|
389
|
+
self._source_is_output = bool(source_is_output)
|
|
390
|
+
self._state: dict[str, DocumentState] = {}
|
|
391
|
+
self._loading = False
|
|
392
|
+
self._index = 0
|
|
393
|
+
self._closed = False
|
|
394
|
+
self._reviewing_flagged = False
|
|
395
|
+
self._autosave = AutosaveOverlay(self.output_path)
|
|
396
|
+
self._output_fingerprint: FileFingerprint | None = None
|
|
397
|
+
self._output_signature: tuple[int, int] | None = None
|
|
398
|
+
|
|
399
|
+
self.output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
400
|
+
lock_path = self.output_path.with_name(f".{self.output_path.name}.lock")
|
|
401
|
+
self._output_lock = FileLock(lock_path)
|
|
402
|
+
|
|
403
|
+
try:
|
|
404
|
+
self._output_lock.acquire(timeout=0)
|
|
405
|
+
except Timeout as exc:
|
|
406
|
+
self._closed = True
|
|
407
|
+
raise RuntimeError(
|
|
408
|
+
f"{self.output_path} is already in use by another spanmark session"
|
|
409
|
+
) from exc
|
|
410
|
+
|
|
411
|
+
try:
|
|
412
|
+
if not self._source_is_output and self.output_path.exists():
|
|
413
|
+
raise FileExistsError(
|
|
414
|
+
f"{self.output_path} already exists. To resume it, pass that "
|
|
415
|
+
"JSONL file as both input and output_path; otherwise choose a "
|
|
416
|
+
"new output_path."
|
|
417
|
+
)
|
|
418
|
+
if not self._source_is_output and self._autosave.exists:
|
|
419
|
+
raise RuntimeError(
|
|
420
|
+
f"Recovery autosave {self._autosave.path} exists without a "
|
|
421
|
+
f"resumable output at {self.output_path}. Move or remove the "
|
|
422
|
+
"orphaned autosave before starting a new session."
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
if self._source_is_output:
|
|
426
|
+
if not isinstance(self._source, JsonlDocumentSource):
|
|
427
|
+
raise RuntimeError("Only a JSONL source can be its own output")
|
|
428
|
+
self._source.refresh()
|
|
429
|
+
self._set_output_identity(fingerprint_file(self.output_path))
|
|
430
|
+
|
|
431
|
+
self._load_states()
|
|
432
|
+
if self._source_is_output and self._autosave.exists:
|
|
433
|
+
self._recover_autosave()
|
|
434
|
+
|
|
435
|
+
self._index = self._resume_index()
|
|
436
|
+
self._decision_history = self._resume_decision_history()
|
|
437
|
+
|
|
438
|
+
self.widget = SpanmarkWidget(
|
|
439
|
+
labels=list(self.labels),
|
|
440
|
+
trim_whitespace=trim_whitespace,
|
|
441
|
+
allow_overlaps=self.allow_overlaps,
|
|
442
|
+
)
|
|
443
|
+
self.widget.observe(self._on_spans, names="spans")
|
|
444
|
+
self.widget.observe(self._on_flagged, names="flagged")
|
|
445
|
+
self.widget.observe(self._on_event, names="event")
|
|
446
|
+
|
|
447
|
+
self._load_current()
|
|
448
|
+
|
|
449
|
+
# A new session starts with a complete JSONL checkpoint. On resume,
|
|
450
|
+
# this also folds any recovered autosave overlay back into the file.
|
|
451
|
+
self._checkpoint()
|
|
452
|
+
except BaseException:
|
|
453
|
+
self._release_lock()
|
|
454
|
+
raise
|
|
455
|
+
|
|
456
|
+
def _load_states(self) -> None:
|
|
457
|
+
for index, expected_id in enumerate(self._source.ids):
|
|
458
|
+
raw = self._source.record(index)
|
|
459
|
+
document = normalize_document(raw, index)
|
|
460
|
+
if document.id != expected_id:
|
|
461
|
+
raise RuntimeError(
|
|
462
|
+
f"Document source index expected {expected_id!r}, "
|
|
463
|
+
f"found {document.id!r}"
|
|
464
|
+
)
|
|
465
|
+
self._state[document.id] = self._state_from_record(document, raw)
|
|
466
|
+
|
|
467
|
+
def _state_from_record(
|
|
468
|
+
self,
|
|
469
|
+
document: Document,
|
|
470
|
+
raw: Mapping[str, Any],
|
|
471
|
+
) -> DocumentState:
|
|
472
|
+
suggestion_spans = tuple(
|
|
473
|
+
suggestion_to_working_span(suggestion, index)
|
|
474
|
+
for index, suggestion in enumerate(document.suggestions)
|
|
475
|
+
)
|
|
476
|
+
suggestion_spans = validate_working_spans(
|
|
477
|
+
suggestion_spans,
|
|
478
|
+
text=document.text,
|
|
479
|
+
labels=self.labels,
|
|
480
|
+
allow_overlaps=self.allow_overlaps,
|
|
481
|
+
)
|
|
482
|
+
|
|
483
|
+
annotation = raw.get("annotation")
|
|
484
|
+
if annotation is None:
|
|
485
|
+
return DocumentState(spans=suggestion_spans)
|
|
486
|
+
if not isinstance(annotation, Mapping):
|
|
487
|
+
raise TypeError(f"Document {document.id!r} 'annotation' must be an object")
|
|
488
|
+
|
|
489
|
+
raw_answer = annotation.get("answer")
|
|
490
|
+
if raw_answer is None:
|
|
491
|
+
answer = ""
|
|
492
|
+
elif isinstance(raw_answer, str) and raw_answer in self.ANSWERS:
|
|
493
|
+
answer = raw_answer
|
|
494
|
+
else:
|
|
495
|
+
raise ValueError(
|
|
496
|
+
f"Document {document.id!r} annotation answer must be one of "
|
|
497
|
+
f"{sorted(self.ANSWERS)} or null"
|
|
498
|
+
)
|
|
499
|
+
|
|
500
|
+
raw_flagged = annotation.get("flagged", False)
|
|
501
|
+
if not isinstance(raw_flagged, bool):
|
|
502
|
+
raise TypeError(
|
|
503
|
+
f"Document {document.id!r} annotation flagged must be a boolean"
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
raw_spans = annotation.get("spans")
|
|
507
|
+
if raw_spans is None:
|
|
508
|
+
if answer or raw_flagged:
|
|
509
|
+
raise ValueError(
|
|
510
|
+
f"Document {document.id!r} annotation spans cannot be null "
|
|
511
|
+
"once a decision or flag has been recorded"
|
|
512
|
+
)
|
|
513
|
+
return DocumentState(spans=suggestion_spans)
|
|
514
|
+
|
|
515
|
+
if isinstance(raw_spans, (str, bytes, Mapping)) or not isinstance(
|
|
516
|
+
raw_spans, Sequence
|
|
517
|
+
):
|
|
518
|
+
raise TypeError(
|
|
519
|
+
f"Document {document.id!r} annotation spans must be a list or null"
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
persisted_spans = cast(
|
|
523
|
+
Sequence[Mapping[str, Any] | WorkingSpan],
|
|
524
|
+
raw_spans,
|
|
525
|
+
)
|
|
526
|
+
spans = validate_working_spans(
|
|
527
|
+
persisted_spans,
|
|
528
|
+
text=document.text,
|
|
529
|
+
labels=self.labels,
|
|
530
|
+
allow_overlaps=self.allow_overlaps,
|
|
531
|
+
)
|
|
532
|
+
return DocumentState(
|
|
533
|
+
spans=spans,
|
|
534
|
+
answer=answer,
|
|
535
|
+
flagged=raw_flagged,
|
|
536
|
+
materialized=True,
|
|
537
|
+
)
|
|
538
|
+
|
|
539
|
+
def _recover_autosave(self) -> None:
|
|
540
|
+
recovery = self._autosave.recover()
|
|
541
|
+
if recovery is None:
|
|
542
|
+
return
|
|
543
|
+
|
|
544
|
+
current_fingerprint = self._output_fingerprint
|
|
545
|
+
if current_fingerprint is None:
|
|
546
|
+
raise RuntimeError("Cannot recover autosave without a JSONL checkpoint")
|
|
547
|
+
|
|
548
|
+
recovered_states: dict[str, DocumentState] = {}
|
|
549
|
+
for document_id, annotation in recovery.annotations.items():
|
|
550
|
+
if document_id not in self._state:
|
|
551
|
+
raise RuntimeError(
|
|
552
|
+
f"Autosave overlay {self._autosave.path} refers to unknown "
|
|
553
|
+
f"document {document_id!r}"
|
|
554
|
+
)
|
|
555
|
+
index = self._source.index_of(document_id)
|
|
556
|
+
document = self._source[index]
|
|
557
|
+
recovered_states[document_id] = self._state_from_record(
|
|
558
|
+
document,
|
|
559
|
+
{"annotation": annotation},
|
|
560
|
+
)
|
|
561
|
+
|
|
562
|
+
if recovery.base == current_fingerprint:
|
|
563
|
+
self._state.update(recovered_states)
|
|
564
|
+
return
|
|
565
|
+
|
|
566
|
+
if all(
|
|
567
|
+
self._state[document_id] == state
|
|
568
|
+
for document_id, state in recovered_states.items()
|
|
569
|
+
):
|
|
570
|
+
# A checkpoint may have been atomically replaced before overlay cleanup.
|
|
571
|
+
# Discarding the stale overlay is safe only because every overlay entry
|
|
572
|
+
# is a complete per-document annotation state, not a patch: if every
|
|
573
|
+
# recovered state is already present in the JSONL, the overlay contains
|
|
574
|
+
# no additional information that could be lost. If overlay records ever
|
|
575
|
+
# become partial updates, this recovery rule must be redesigned.
|
|
576
|
+
return
|
|
577
|
+
|
|
578
|
+
raise RuntimeError(
|
|
579
|
+
f"Autosave overlay {self._autosave.path} was based on a different "
|
|
580
|
+
"JSONL checkpoint and cannot be safely recovered. The output may have "
|
|
581
|
+
"been modified outside spanmark."
|
|
582
|
+
)
|
|
583
|
+
|
|
584
|
+
def _current_doc(self) -> Document:
|
|
585
|
+
return self._source[self._index]
|
|
586
|
+
|
|
587
|
+
def _decided_count(self) -> int:
|
|
588
|
+
return sum(state.answer in self.ANSWERS for state in self._state.values())
|
|
589
|
+
|
|
590
|
+
def _answer_count(self, answer: str) -> int:
|
|
591
|
+
return sum(state.answer == answer for state in self._state.values())
|
|
592
|
+
|
|
593
|
+
def _flagged_count(self) -> int:
|
|
594
|
+
return sum(state.flagged for state in self._state.values())
|
|
595
|
+
|
|
596
|
+
def _is_complete(self) -> bool:
|
|
597
|
+
return self._decided_count() == len(self._source)
|
|
598
|
+
|
|
599
|
+
def _sync_summary_traits(self) -> None:
|
|
600
|
+
self.widget.decided_count = self._decided_count()
|
|
601
|
+
self.widget.accept_count = self._answer_count("accept")
|
|
602
|
+
self.widget.reject_count = self._answer_count("reject")
|
|
603
|
+
self.widget.ignore_count = self._answer_count("ignore")
|
|
604
|
+
self.widget.flagged_count = self._flagged_count()
|
|
605
|
+
self.widget.complete = self._is_complete()
|
|
606
|
+
self.widget.reviewing_flagged = self._reviewing_flagged
|
|
607
|
+
self.widget.can_undo = (
|
|
608
|
+
bool(self._decision_history) and not self._reviewing_flagged
|
|
609
|
+
)
|
|
610
|
+
|
|
611
|
+
def _resume_index(self) -> int:
|
|
612
|
+
for index, document_id in enumerate(self._source.ids):
|
|
613
|
+
if self._state[document_id].answer not in self.ANSWERS:
|
|
614
|
+
return index
|
|
615
|
+
return len(self._source) - 1
|
|
616
|
+
|
|
617
|
+
def _resume_decision_history(self) -> list[str]:
|
|
618
|
+
current_id = self._source.ids[self._index]
|
|
619
|
+
current_decided = self._state[current_id].answer in self.ANSWERS
|
|
620
|
+
stop = self._index + 1 if current_decided else self._index
|
|
621
|
+
|
|
622
|
+
# JSONL stores decisions but not action chronology. Reconstructing history
|
|
623
|
+
# in dataset order matches the normal one-way pass, but cannot reproduce
|
|
624
|
+
# arbitrary pre-close edit/redecision order exactly.
|
|
625
|
+
return [
|
|
626
|
+
document_id
|
|
627
|
+
for document_id in self._source.ids[:stop]
|
|
628
|
+
if self._state[document_id].answer in self.ANSWERS
|
|
629
|
+
]
|
|
630
|
+
|
|
631
|
+
def _next_undecided_index(self, after: int) -> int | None:
|
|
632
|
+
for index in range(after + 1, len(self._source)):
|
|
633
|
+
document_id = self._source.ids[index]
|
|
634
|
+
if self._state[document_id].answer not in self.ANSWERS:
|
|
635
|
+
return index
|
|
636
|
+
return None
|
|
637
|
+
|
|
638
|
+
def _first_flagged_index(self) -> int | None:
|
|
639
|
+
for index, document_id in enumerate(self._source.ids):
|
|
640
|
+
if self._state[document_id].flagged:
|
|
641
|
+
return index
|
|
642
|
+
return None
|
|
643
|
+
|
|
644
|
+
def _start_flagged_review(self) -> None:
|
|
645
|
+
if not self._is_complete():
|
|
646
|
+
self.widget.status = (
|
|
647
|
+
"Finish the main annotation pass before reviewing flags."
|
|
648
|
+
)
|
|
649
|
+
return
|
|
650
|
+
|
|
651
|
+
first_index = self._first_flagged_index()
|
|
652
|
+
if first_index is None:
|
|
653
|
+
self.widget.status = "There are no flagged examples to review."
|
|
654
|
+
return
|
|
655
|
+
|
|
656
|
+
self._reviewing_flagged = True
|
|
657
|
+
self._index = first_index
|
|
658
|
+
self._load_current()
|
|
659
|
+
self.widget.status = (
|
|
660
|
+
"Review flagged examples — clear Flag when this item is resolved."
|
|
661
|
+
)
|
|
662
|
+
|
|
663
|
+
def _load_current(self) -> None:
|
|
664
|
+
document = self._current_doc()
|
|
665
|
+
self._set_widget_document(document, self._state[document.id])
|
|
666
|
+
|
|
667
|
+
def _set_widget_document(
|
|
668
|
+
self,
|
|
669
|
+
document: Document,
|
|
670
|
+
state: DocumentState,
|
|
671
|
+
*,
|
|
672
|
+
status: str = "",
|
|
673
|
+
) -> None:
|
|
674
|
+
self._loading = True
|
|
675
|
+
try:
|
|
676
|
+
self.widget.doc_id = document.id
|
|
677
|
+
self.widget.text = document.text
|
|
678
|
+
self.widget.index = self._index
|
|
679
|
+
self.widget.total = len(self._source)
|
|
680
|
+
self._set_widget_annotation_state(state, status=status)
|
|
681
|
+
finally:
|
|
682
|
+
self._loading = False
|
|
683
|
+
|
|
684
|
+
def _set_widget_annotation_state(
|
|
685
|
+
self,
|
|
686
|
+
state: DocumentState,
|
|
687
|
+
*,
|
|
688
|
+
status: str = "",
|
|
689
|
+
) -> None:
|
|
690
|
+
self.widget.spans = [working_span_to_dict(span) for span in state.spans]
|
|
691
|
+
self.widget.answer = state.answer
|
|
692
|
+
self.widget.flagged = state.flagged
|
|
693
|
+
self._sync_summary_traits()
|
|
694
|
+
self.widget.status = status
|
|
695
|
+
|
|
696
|
+
def _restore_current_annotation_state(self, *, status: str) -> None:
|
|
697
|
+
self._loading = True
|
|
698
|
+
try:
|
|
699
|
+
self._set_widget_annotation_state(
|
|
700
|
+
self._state[self.current_id],
|
|
701
|
+
status=status,
|
|
702
|
+
)
|
|
703
|
+
finally:
|
|
704
|
+
self._loading = False
|
|
705
|
+
|
|
706
|
+
def _on_spans(self, change: Mapping[str, Any]) -> None:
|
|
707
|
+
if self._loading or self._closed:
|
|
708
|
+
return
|
|
709
|
+
|
|
710
|
+
doc_id = self.current_id
|
|
711
|
+
previous_state = self._state[doc_id]
|
|
712
|
+
|
|
713
|
+
try:
|
|
714
|
+
spans = validate_working_spans(
|
|
715
|
+
change["new"],
|
|
716
|
+
text=self.widget.text,
|
|
717
|
+
labels=self.labels,
|
|
718
|
+
allow_overlaps=self.allow_overlaps,
|
|
719
|
+
)
|
|
720
|
+
except Exception as exc:
|
|
721
|
+
self._restore_current_annotation_state(
|
|
722
|
+
status=f"Span edit rejected: {exc}",
|
|
723
|
+
)
|
|
724
|
+
return
|
|
725
|
+
|
|
726
|
+
state = replace(
|
|
727
|
+
previous_state,
|
|
728
|
+
spans=spans,
|
|
729
|
+
materialized=True,
|
|
730
|
+
)
|
|
731
|
+
decision_history = self._decision_history
|
|
732
|
+
status = ""
|
|
733
|
+
|
|
734
|
+
# A decision certifies the exact current span set. Any edit invalidates
|
|
735
|
+
# the previous decision while preserving Flag.
|
|
736
|
+
if state.answer in self.ANSWERS:
|
|
737
|
+
state = replace(state, answer="")
|
|
738
|
+
decision_history = [
|
|
739
|
+
saved_id for saved_id in self._decision_history if saved_id != doc_id
|
|
740
|
+
]
|
|
741
|
+
if self._reviewing_flagged:
|
|
742
|
+
status = "Span edited — choose Accept, Reject, or Ignore before clearing Flag."
|
|
743
|
+
else:
|
|
744
|
+
status = "Span edited — previous decision cleared."
|
|
745
|
+
|
|
746
|
+
try:
|
|
747
|
+
self._append_state(doc_id, state)
|
|
748
|
+
except Exception as exc:
|
|
749
|
+
self._restore_current_annotation_state(
|
|
750
|
+
status=f"Could not save span edit: {exc}",
|
|
751
|
+
)
|
|
752
|
+
return
|
|
753
|
+
|
|
754
|
+
self._state[doc_id] = state
|
|
755
|
+
self._decision_history = decision_history
|
|
756
|
+
self._loading = True
|
|
757
|
+
try:
|
|
758
|
+
self._set_widget_annotation_state(state, status=status)
|
|
759
|
+
finally:
|
|
760
|
+
self._loading = False
|
|
761
|
+
|
|
762
|
+
warning = self._checkpoint_warning_if_large()
|
|
763
|
+
if warning is not None:
|
|
764
|
+
self.widget.status = warning
|
|
765
|
+
|
|
766
|
+
def _on_flagged(self, change: Mapping[str, Any]) -> None:
|
|
767
|
+
if self._loading or self._closed:
|
|
768
|
+
return
|
|
769
|
+
|
|
770
|
+
previous_state = self._state[self.current_id]
|
|
771
|
+
try:
|
|
772
|
+
self.flag(bool(change["new"]))
|
|
773
|
+
except Exception as exc:
|
|
774
|
+
self._loading = True
|
|
775
|
+
try:
|
|
776
|
+
self.widget.flagged = previous_state.flagged
|
|
777
|
+
self._sync_summary_traits()
|
|
778
|
+
self.widget.status = f"Could not save Flag change: {exc}"
|
|
779
|
+
finally:
|
|
780
|
+
self._loading = False
|
|
781
|
+
|
|
782
|
+
def _on_event(self, change: Mapping[str, Any]) -> None:
|
|
783
|
+
if self._loading or self._closed:
|
|
784
|
+
return
|
|
785
|
+
|
|
786
|
+
event = change["new"] or {}
|
|
787
|
+
kind = event.get("type")
|
|
788
|
+
|
|
789
|
+
try:
|
|
790
|
+
if kind == "decision":
|
|
791
|
+
self.decide(str(event.get("answer") or ""))
|
|
792
|
+
elif kind == "undo_decision":
|
|
793
|
+
self.undo_decision()
|
|
794
|
+
elif kind == "review_flagged":
|
|
795
|
+
self._start_flagged_review()
|
|
796
|
+
except Exception as exc:
|
|
797
|
+
self.widget.status = f"Could not apply action: {exc}"
|
|
798
|
+
|
|
799
|
+
def _append_state(self, document_id: str, state: DocumentState) -> None:
|
|
800
|
+
self._check_source_unchanged()
|
|
801
|
+
self._check_output_unchanged()
|
|
802
|
+
|
|
803
|
+
base = self._output_fingerprint
|
|
804
|
+
if base is None:
|
|
805
|
+
raise RuntimeError("No JSONL checkpoint is available for autosave")
|
|
806
|
+
|
|
807
|
+
self._autosave.append(
|
|
808
|
+
base=base,
|
|
809
|
+
document_id=document_id,
|
|
810
|
+
annotation=self._annotation_for_state(state),
|
|
811
|
+
)
|
|
812
|
+
|
|
813
|
+
def _checkpoint(self) -> Path:
|
|
814
|
+
self._check_source_unchanged()
|
|
815
|
+
self._check_output_unchanged()
|
|
816
|
+
|
|
817
|
+
result, fingerprint = atomic_write_jsonl_with_fingerprint(
|
|
818
|
+
self.output_path,
|
|
819
|
+
self._iter_records(),
|
|
820
|
+
)
|
|
821
|
+
|
|
822
|
+
if self._source_is_output:
|
|
823
|
+
if not isinstance(self._source, JsonlDocumentSource):
|
|
824
|
+
raise RuntimeError("Only a JSONL source can be its own output")
|
|
825
|
+
self._source.refresh()
|
|
826
|
+
|
|
827
|
+
self._set_output_identity(fingerprint)
|
|
828
|
+
self._autosave.discard()
|
|
829
|
+
return result
|
|
830
|
+
|
|
831
|
+
def _checkpoint_warning_if_large(self) -> str | None:
|
|
832
|
+
if self._autosave.size < AUTOSAVE_CHECKPOINT_BYTES:
|
|
833
|
+
return None
|
|
834
|
+
|
|
835
|
+
try:
|
|
836
|
+
self._checkpoint()
|
|
837
|
+
except Exception as exc:
|
|
838
|
+
return f"Autosaved, but JSONL checkpoint failed: {exc}"
|
|
839
|
+
return None
|
|
840
|
+
|
|
841
|
+
def _checkpoint_with_status(self, status: str) -> None:
|
|
842
|
+
try:
|
|
843
|
+
self._checkpoint()
|
|
844
|
+
except Exception as exc:
|
|
845
|
+
self.widget.status = (
|
|
846
|
+
f"{status} Autosaved, but JSONL checkpoint failed: {exc}"
|
|
847
|
+
)
|
|
848
|
+
else:
|
|
849
|
+
self.widget.status = status
|
|
850
|
+
|
|
851
|
+
def _check_source_unchanged(self) -> None:
|
|
852
|
+
if isinstance(self._source, JsonlDocumentSource):
|
|
853
|
+
self._source.check_unchanged()
|
|
854
|
+
|
|
855
|
+
def _check_output_unchanged(self) -> None:
|
|
856
|
+
if self._output_signature is None:
|
|
857
|
+
return
|
|
858
|
+
|
|
859
|
+
stat = self.output_path.stat()
|
|
860
|
+
current = stat.st_size, stat.st_mtime_ns
|
|
861
|
+
if current != self._output_signature:
|
|
862
|
+
raise RuntimeError(
|
|
863
|
+
f"JSONL output {self.output_path} changed while the session was open"
|
|
864
|
+
)
|
|
865
|
+
|
|
866
|
+
def _set_output_identity(self, fingerprint: FileFingerprint) -> None:
|
|
867
|
+
stat = self.output_path.stat()
|
|
868
|
+
self._output_fingerprint = fingerprint
|
|
869
|
+
self._output_signature = stat.st_size, stat.st_mtime_ns
|
|
870
|
+
|
|
871
|
+
def _annotation_for_state(self, state: DocumentState) -> dict[str, Any]:
|
|
872
|
+
return {
|
|
873
|
+
"spans": (
|
|
874
|
+
[annotation_span_to_dict(span) for span in state.spans]
|
|
875
|
+
if state.materialized
|
|
876
|
+
else None
|
|
877
|
+
),
|
|
878
|
+
"answer": state.answer or None,
|
|
879
|
+
"flagged": state.flagged,
|
|
880
|
+
}
|
|
881
|
+
|
|
882
|
+
def _iter_records(self) -> Iterator[dict[str, Any]]:
|
|
883
|
+
for index, document_id in enumerate(self._source.ids):
|
|
884
|
+
record = dict(self._source.record(index))
|
|
885
|
+
record["annotation"] = self._annotation_for_state(self._state[document_id])
|
|
886
|
+
yield record
|
|
887
|
+
|
|
888
|
+
def _release_lock(self) -> None:
|
|
889
|
+
if self._closed:
|
|
890
|
+
return
|
|
891
|
+
self._output_lock.release()
|
|
892
|
+
self._closed = True
|
|
893
|
+
|
|
894
|
+
def _ensure_open(self) -> None:
|
|
895
|
+
if self._closed:
|
|
896
|
+
raise RuntimeError("This spanmark session is closed")
|