agent2learn 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. agent2learn/__init__.py +3 -0
  2. agent2learn/_release.py +19 -0
  3. agent2learn/aipolicy.py +182 -0
  4. agent2learn/api.py +590 -0
  5. agent2learn/audit.py +358 -0
  6. agent2learn/auth/__init__.py +282 -0
  7. agent2learn/auth/cdp.py +1067 -0
  8. agent2learn/auth/paste.py +378 -0
  9. agent2learn/calendar.py +525 -0
  10. agent2learn/calibrate.py +347 -0
  11. agent2learn/check.py +1091 -0
  12. agent2learn/cli.py +2039 -0
  13. agent2learn/clock.py +39 -0
  14. agent2learn/config.py +205 -0
  15. agent2learn/console.py +229 -0
  16. agent2learn/convert.py +1223 -0
  17. agent2learn/doctor.py +1167 -0
  18. agent2learn/errors.py +32 -0
  19. agent2learn/ground.py +735 -0
  20. agent2learn/index.py +614 -0
  21. agent2learn/ingest.py +3229 -0
  22. agent2learn/locations.py +247 -0
  23. agent2learn/outlines.py +754 -0
  24. agent2learn/paths.py +683 -0
  25. agent2learn/pipeline.py +392 -0
  26. agent2learn/privacy.py +1123 -0
  27. agent2learn/schools/__init__.py +29 -0
  28. agent2learn/schools/_base.py +194 -0
  29. agent2learn/schools/generic.py +78 -0
  30. agent2learn/schools/uwaterloo.py +66 -0
  31. agent2learn/session.py +373 -0
  32. agent2learn/skills.py +1081 -0
  33. agent2learn/snapshot.py +399 -0
  34. agent2learn/submit.py +1047 -0
  35. agent2learn/transactions.py +157 -0
  36. agent2learn/upgrade.py +288 -0
  37. agent2learn/vault.py +1134 -0
  38. agent2learn-0.1.2.data/data/a2l-coursework/SKILL.md +52 -0
  39. agent2learn-0.1.2.data/data/a2l-setup/SKILL.md +27 -0
  40. agent2learn-0.1.2.data/data/a2l-study/SKILL.md +27 -0
  41. agent2learn-0.1.2.data/data/a2l-sync/SKILL.md +30 -0
  42. agent2learn-0.1.2.dist-info/METADATA +186 -0
  43. agent2learn-0.1.2.dist-info/RECORD +46 -0
  44. agent2learn-0.1.2.dist-info/WHEEL +4 -0
  45. agent2learn-0.1.2.dist-info/entry_points.txt +3 -0
  46. agent2learn-0.1.2.dist-info/licenses/LICENSE +202 -0
agent2learn/vault.py ADDED
@@ -0,0 +1,1134 @@
1
+ """Portable vault state, structured manifests, and revision-safe storage.
2
+
3
+ The manifest is the source of truth for course identity and provenance. It stores only
4
+ vault-relative POSIX paths, and every path is resolved against the current vault root when it is
5
+ used. A filename or display title is never treated as a source identity.
6
+
7
+ Revision preservation is deliberately a separate operation from installing a new source. The
8
+ caller verifies and preserves the old revision first, then performs its own atomic materialized
9
+ file replacement and manifest update. This keeps a failed history write from pretending that a
10
+ revision was safely archived.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import os
17
+ import re
18
+ import stat
19
+ import subprocess
20
+ import sys
21
+ import tempfile
22
+ import unicodedata
23
+ from collections.abc import Callable, Collection, Mapping
24
+ from contextlib import suppress
25
+ from dataclasses import dataclass, field
26
+ from datetime import UTC, datetime, timedelta
27
+ from hashlib import sha256
28
+ from pathlib import Path, PurePosixPath
29
+ from typing import Any, BinaryIO, cast
30
+
31
+ from agent2learn import paths
32
+ from agent2learn.errors import A2LError
33
+
34
+ SCHEMA_VERSION = 1
35
+ MIGRATIONS: dict[int, Callable[[Vault], None]] = {}
36
+
37
+ _SHA256 = re.compile(r"^[0-9a-f]{64}$")
38
+ _WINDOWS_ABSOLUTE = re.compile(r"^[A-Za-z]:[\\/]")
39
+ _KEY_PART = re.compile(r"^[^:\\/:\s]+$")
40
+ _MANIFEST_KEYS = frozenset(
41
+ {"path", "sha256", "source_id", "etag", "last_modified", "size", "fetched_at", "derived"}
42
+ )
43
+ _DERIVED_REQUIRED_KEYS = frozenset(
44
+ {"path", "sha256", "source_sha256", "tool", "tool_version", "created_at"}
45
+ )
46
+ _DERIVED_OPTIONAL_KEYS = frozenset({"ocr_words_per_page", "page_coverage"})
47
+ _DERIVED_KEYS = _DERIVED_REQUIRED_KEYS | _DERIVED_OPTIONAL_KEYS
48
+ _GIT_CONFIRMATION = "I UNDERSTAND THIS VAULT IS IN A GIT WORKTREE"
49
+ _COPY_CHUNK_SIZE = 1024 * 1024
50
+
51
+
52
+ @dataclass(frozen=True)
53
+ class DerivedArtifact:
54
+ """A generated artifact tied to one exact source revision."""
55
+
56
+ path: str
57
+ sha256: str
58
+ source_sha256: str
59
+ tool: str
60
+ tool_version: str
61
+ created_at: str
62
+ ocr_words_per_page: int | None = None
63
+ page_coverage: tuple[Mapping[str, object], ...] = ()
64
+
65
+
66
+ @dataclass(frozen=True)
67
+ class ManifestEntry:
68
+ """A structured record for one stable remote source identity."""
69
+
70
+ path: str
71
+ sha256: str
72
+ source_id: str
73
+ etag: str | None
74
+ last_modified: str | None
75
+ size: int
76
+ fetched_at: str
77
+ derived: Mapping[str, DerivedArtifact] = field(default_factory=dict)
78
+
79
+
80
+ class Vault:
81
+ """Access a portable Agent2Learn vault rooted at ``root``."""
82
+
83
+ root: Path
84
+ _entries: dict[str, ManifestEntry]
85
+ _loaded: bool
86
+ _state_override: Path | None
87
+ _schema_migration_active: bool
88
+
89
+ def __init__(self, root: Path) -> None:
90
+ self.root = Path(root).expanduser().resolve()
91
+ self._entries = {}
92
+ self._loaded = False
93
+ self._state_override = None
94
+ self._schema_migration_active = False
95
+
96
+ def state(self) -> Path:
97
+ """Return the vault-scoped state directory."""
98
+
99
+ state = self._state_override if self._state_override is not None else self.root / ".a2l"
100
+ if paths.is_link(state):
101
+ raise A2LError("vault state directory must not be a symlink")
102
+ return state
103
+
104
+ def history_bucket(self, source_key: str) -> Path:
105
+ """Return and create the opaque history bucket for a canonical source key."""
106
+
107
+ _validate_source_key(source_key)
108
+ check_schema(self)
109
+ bucket = self.state() / "history" / sha256(source_key.encode("utf-8")).hexdigest()
110
+ paths.ensure_dir(bucket, root=self.root)
111
+ return bucket
112
+
113
+ def manifest(self) -> dict[str, ManifestEntry]:
114
+ """Load, validate, and return a copy of the structured manifest."""
115
+
116
+ check_schema(self)
117
+ if self._loaded:
118
+ return dict(self._entries)
119
+
120
+ destination = self.state() / "manifest.json"
121
+ if paths.is_link(destination):
122
+ raise A2LError("manifest must not be a symlink")
123
+ raw = _read_json_object(destination, "manifest")
124
+ if raw is None:
125
+ self._entries = {}
126
+ self._loaded = True
127
+ return {}
128
+ schema_version = raw.get("schema_version")
129
+ if isinstance(schema_version, bool) or not isinstance(schema_version, int):
130
+ raise A2LError("manifest schema_version must be an integer")
131
+ if schema_version != SCHEMA_VERSION:
132
+ raise A2LError(
133
+ f"manifest schema version {schema_version} does not match tool schema "
134
+ f"{SCHEMA_VERSION}"
135
+ )
136
+
137
+ entries_raw = raw.get("entries")
138
+ if not isinstance(entries_raw, dict):
139
+ raise A2LError("manifest entries must be an object")
140
+
141
+ entries: dict[str, ManifestEntry] = {}
142
+ for key, raw_entry in entries_raw.items():
143
+ if not isinstance(key, str):
144
+ raise A2LError("manifest entry keys must be canonical source keys")
145
+ entries[key] = _entry_from_json(key, raw_entry)
146
+
147
+ unknown = set(raw) - {"schema_version", "entries"}
148
+ if unknown:
149
+ raise A2LError(f"manifest contains unknown top-level fields: {sorted(unknown)!r}")
150
+
151
+ self._entries = entries
152
+ self._loaded = True
153
+ return dict(entries)
154
+
155
+ def entry(self, key: str) -> ManifestEntry | None:
156
+ """Return one manifest entry by canonical source key."""
157
+
158
+ _validate_source_key(key)
159
+ return self.manifest().get(key)
160
+
161
+ def materialized(self, entry: ManifestEntry) -> Path:
162
+ """Resolve a validated manifest path against this vault's current root."""
163
+
164
+ validated = _validate_entry(entry)
165
+ return self._materialized_path(validated.path)
166
+
167
+ def owns_derived_path(self, key: str, relative: str, *, name: str = "markdown") -> bool:
168
+ entries = self.manifest()
169
+ entry = entries.get(key)
170
+ artifact = entry.derived.get(name) if entry is not None else None
171
+ if artifact is None or artifact.path != relative:
172
+ return False
173
+ wanted = unicodedata.normalize("NFC", relative).casefold()
174
+ for source_key, source in entries.items():
175
+ if unicodedata.normalize("NFC", source.path).casefold() == wanted:
176
+ return False
177
+ for artifact_name, other in source.derived.items():
178
+ if (source_key, artifact_name) != (key, name) and (
179
+ unicodedata.normalize("NFC", other.path).casefold() == wanted
180
+ ):
181
+ return False
182
+ return True
183
+
184
+ def derived_destination(self, key: str, preferred: Path, *, name: str = "markdown") -> Path:
185
+ _validate_source_key(key)
186
+ entries = self.manifest()
187
+ entry = entries.get(key)
188
+ artifact = entry.derived.get(name) if entry is not None else None
189
+ if artifact is not None and self.owns_derived_path(key, artifact.path, name=name):
190
+ return self._materialized_path(artifact.path)
191
+ if paths.has_link_component(preferred.parent, root=self.root):
192
+ raise A2LError("derived destination is outside the trusted vault")
193
+ return paths.unique_path(preferred, reserved=self.claimed_paths())
194
+
195
+ def claimed_paths(self) -> tuple[Path, ...]:
196
+ return tuple(
197
+ self.root / PurePosixPath(relative)
198
+ for source in self.manifest().values()
199
+ for relative in (source.path, *(value.path for value in source.derived.values()))
200
+ )
201
+
202
+ def mark(self, key: str, entry: ManifestEntry) -> None:
203
+ """Validate and stage a manifest entry for the next atomic save."""
204
+
205
+ _validate_source_key(key)
206
+ if not self._loaded:
207
+ self.manifest()
208
+ checked = _validate_entry(entry)
209
+ _validate_source_id(key, checked.source_id)
210
+ self._entries[key] = checked
211
+
212
+ def preserve_revision(self, key: str, *, changed_at: datetime) -> Path | None:
213
+ """Atomically preserve the verified current source before replacing its bytes.
214
+
215
+ ``None`` means the source is not materialized yet. A materialized source whose bytes no
216
+ longer match the manifest raises an integrity gap instead of inventing a revision from
217
+ untrusted local data.
218
+ """
219
+
220
+ _validate_source_key(key)
221
+ changed_utc = _utc_datetime(changed_at, "changed_at")
222
+ current = self.entry(key)
223
+ if current is None:
224
+ return None
225
+
226
+ source = self.materialized(current)
227
+ revision_source = _allocate_revision_directory(
228
+ self.history_bucket(key), changed_utc.strftime("%Y%m%dT%H%M%SZ"), root=self.root
229
+ )
230
+ try:
231
+ saved_source = revision_source / PurePosixPath(current.path).name
232
+ if not _copy_verified(
233
+ source,
234
+ saved_source,
235
+ root=self.root,
236
+ expected_sha256=current.sha256,
237
+ expected_size=current.size,
238
+ ):
239
+ paths.remove_tree(revision_source)
240
+ return None
241
+
242
+ derived_metadata: dict[str, dict[str, object]] = {}
243
+ for name, artifact in current.derived.items():
244
+ artifact_path = self._materialized_path(artifact.path)
245
+ artifact_destination = (
246
+ revision_source / "derived" / PurePosixPath(artifact.path).name
247
+ )
248
+ if not _copy_verified(
249
+ artifact_path,
250
+ artifact_destination,
251
+ root=self.root,
252
+ expected_sha256=None,
253
+ expected_size=None,
254
+ ):
255
+ derived_metadata[name] = {
256
+ "path": artifact.path,
257
+ "sha256": artifact.sha256,
258
+ "source_sha256": artifact.source_sha256,
259
+ "tool": artifact.tool,
260
+ "tool_version": artifact.tool_version,
261
+ "created_at": artifact.created_at,
262
+ "ocr_words_per_page": artifact.ocr_words_per_page,
263
+ "page_coverage": [dict(page) for page in artifact.page_coverage],
264
+ "status": "missing",
265
+ }
266
+ continue
267
+
268
+ actual_sha256, actual_size = _hash_file(artifact_path)
269
+ derived_metadata[name] = {
270
+ "path": artifact.path,
271
+ "sha256": artifact.sha256,
272
+ "source_sha256": artifact.source_sha256,
273
+ "tool": artifact.tool,
274
+ "tool_version": artifact.tool_version,
275
+ "created_at": artifact.created_at,
276
+ "ocr_words_per_page": artifact.ocr_words_per_page,
277
+ "page_coverage": [dict(page) for page in artifact.page_coverage],
278
+ "actual_sha256": actual_sha256,
279
+ "size": actual_size,
280
+ "status": (
281
+ "verified" if actual_sha256 == artifact.sha256 else "local-modification"
282
+ ),
283
+ }
284
+
285
+ metadata = {
286
+ "canonical_key": key,
287
+ "fetched_at": current.fetched_at,
288
+ "new_sha256": None,
289
+ "old_sha256": current.sha256,
290
+ "path": current.path,
291
+ "preserved_at": changed_utc.isoformat().replace("+00:00", "Z"),
292
+ "size": current.size,
293
+ "source_key": key,
294
+ "derived": derived_metadata,
295
+ }
296
+ paths.atomic_write_text(
297
+ revision_source / "revision.json", _canonical_json(metadata), root=self.root
298
+ )
299
+ return saved_source
300
+ except BaseException:
301
+ paths.remove_tree(revision_source, ignore_errors=True)
302
+ raise
303
+
304
+ def save_manifest(self) -> None:
305
+ """Atomically save the canonical manifest and current schema marker."""
306
+
307
+ check_schema(self)
308
+ if not self._loaded:
309
+ self.manifest()
310
+
311
+ payload = {
312
+ "entries": {
313
+ key: _entry_to_json(_validate_source_entry(key, entry))
314
+ for key, entry in sorted(self._entries.items(), key=lambda item: item[0])
315
+ },
316
+ "schema_version": SCHEMA_VERSION,
317
+ }
318
+ paths.atomic_write_text(
319
+ self.state() / "manifest.json", _canonical_json(payload), root=self.root
320
+ )
321
+
322
+ def semesters(self) -> list[Path]:
323
+ """Return direct child term directories that carry semester metadata."""
324
+
325
+ if not paths.long_path(self.root).is_dir():
326
+ return []
327
+ children: list[Path] = []
328
+ with os.scandir(os.fspath(paths.long_path(self.root))) as iterator:
329
+ for entry in iterator:
330
+ child = self.root / entry.name
331
+ if entry.is_dir(follow_symlinks=False) and _regular_file_exists(
332
+ child / "_SEMESTER_METADATA.json", "semester metadata"
333
+ ):
334
+ children.append(child)
335
+ return sorted(children, key=lambda path: path.name)
336
+
337
+ @staticmethod
338
+ def is_vault(p: Path) -> bool:
339
+ """Return whether ``p`` has an Agent2Learn marker."""
340
+
341
+ candidate = Path(p)
342
+ state = candidate / ".a2l"
343
+ if paths.is_link(state):
344
+ return False
345
+ if paths.long_path(state).is_dir():
346
+ return True
347
+ try:
348
+ return _regular_file_exists(candidate / "_SEMESTER_METADATA.json", "semester metadata")
349
+ except A2LError:
350
+ return False
351
+
352
+ @classmethod
353
+ def claim(cls, p: Path, *, allow_suffix: bool = True) -> Path:
354
+ """Claim a vault root without adopting an unrelated directory.
355
+
356
+ ``allow_suffix=False`` is used by consentful onboarding after it has previewed a path.
357
+ A race that occupies that exact path must fail safely instead of silently moving the
358
+ user's approval to a newly allocated sibling.
359
+ """
360
+
361
+ if not isinstance(allow_suffix, bool):
362
+ raise ValueError("allow_suffix must be a boolean")
363
+
364
+ requested = Path(p).expanduser().resolve()
365
+ _refuse_agent2learn_checkout(requested)
366
+ _require_git_confirmation_if_needed(requested)
367
+
368
+ candidate = requested
369
+ suffix = 2
370
+ while True:
371
+ if cls.is_vault(candidate):
372
+ return candidate
373
+ if not _occupied(candidate):
374
+ try:
375
+ paths.long_path(candidate).mkdir(parents=True, exist_ok=False)
376
+ except FileExistsError:
377
+ if not allow_suffix:
378
+ raise A2LError("vault path became occupied after confirmation") from None
379
+ candidate = requested.with_name(f"{requested.name}-{suffix}")
380
+ suffix += 1
381
+ continue
382
+ _write_vault_gitignore(candidate)
383
+ paths.ensure_dir(candidate / ".a2l", root=candidate)
384
+ check_schema(cls(candidate))
385
+ return candidate
386
+ if not allow_suffix:
387
+ raise A2LError("vault path became occupied after confirmation")
388
+ candidate = requested.with_name(f"{requested.name}-{suffix}")
389
+ suffix += 1
390
+
391
+ def _materialized_path(self, relative: str) -> Path:
392
+ _validate_relative_posix(relative, field="path")
393
+ # Preserve the lexical path until the filesystem boundary. Resolving first would erase
394
+ # an intermediate symlink and make a manifest entry capable of redirecting a managed
395
+ # writer outside the trusted vault (or into a linked directory within it).
396
+ candidate = self.root / Path(*PurePosixPath(relative).parts)
397
+ if paths.has_link_component(candidate, root=self.root):
398
+ raise A2LError("manifest path contains a link component")
399
+ try:
400
+ candidate.relative_to(self.root)
401
+ except ValueError as exc:
402
+ raise A2LError("manifest path escapes the vault root") from exc
403
+ return candidate
404
+
405
+
406
+ def check_schema(v: Vault) -> None:
407
+ """Check or migrate a vault schema, refusing to write newer vaults."""
408
+
409
+ if v._schema_migration_active:
410
+ return
411
+
412
+ state = v.state()
413
+ paths.ensure_dir(state, root=v.root)
414
+ version_path = state / "VERSION"
415
+ if not _regular_file_exists(version_path, "vault VERSION"):
416
+ try:
417
+ with os.scandir(os.fspath(paths.long_path(state))) as iterator:
418
+ has_other_state = any(entry.name != "VERSION" for entry in iterator)
419
+ except OSError as exc:
420
+ raise A2LError("vault VERSION is unreadable") from exc
421
+ if has_other_state:
422
+ raise A2LError("vault VERSION is missing from an existing vault state")
423
+ paths.atomic_write_text(version_path, f"{SCHEMA_VERSION}\n", root=v.root)
424
+ return
425
+
426
+ raw_version = _read_text(version_path).strip()
427
+ try:
428
+ version = int(raw_version)
429
+ except ValueError as exc:
430
+ raise A2LError("vault VERSION must contain one integer") from exc
431
+ if version < 0 or str(version) != raw_version:
432
+ raise A2LError("vault VERSION must contain one non-negative integer")
433
+ if version == SCHEMA_VERSION:
434
+ return
435
+ if version > SCHEMA_VERSION:
436
+ raise A2LError(
437
+ f"vault schema {version} is newer than this tool's schema {SCHEMA_VERSION}; "
438
+ "run a2l upgrade"
439
+ )
440
+
441
+ backup = _backup_state(state, version, root=v.root)
442
+ staging_root = paths.plain_path(
443
+ Path(
444
+ tempfile.mkdtemp(
445
+ prefix=".a2l-migration-",
446
+ dir=os.fspath(paths.long_path(state.parent)),
447
+ )
448
+ )
449
+ )
450
+ staged_state = staging_root / ".a2l"
451
+ try:
452
+ _copy_tree(state, staged_state, root=v.root)
453
+ staged_vault = Vault(v.root)
454
+ staged_vault._state_override = staged_state
455
+ staged_vault._schema_migration_active = True
456
+
457
+ current = version
458
+ while current < SCHEMA_VERSION:
459
+ migration = MIGRATIONS.get(current)
460
+ if migration is None:
461
+ raise A2LError(
462
+ f"vault schema {version} is older than {SCHEMA_VERSION} and has no "
463
+ f"registered migration from {current}"
464
+ )
465
+ migration(staged_vault)
466
+ current += 1
467
+
468
+ paths.atomic_write_text(staged_state / "VERSION", f"{SCHEMA_VERSION}\n", root=v.root)
469
+ _install_migrated_state(state, staged_state, backup, root=v.root)
470
+ finally:
471
+ paths.remove_tree(staging_root, ignore_errors=True)
472
+
473
+
474
+ def _entry_from_json(key: str, raw: object) -> ManifestEntry:
475
+ _validate_source_key(key)
476
+ if not isinstance(raw, dict):
477
+ raise A2LError(f"manifest entry for {key!r} must be an object")
478
+ unknown = set(raw) - _MANIFEST_KEYS
479
+ if unknown:
480
+ raise A2LError(f"manifest entry for {key!r} has unknown fields: {sorted(unknown)!r}")
481
+
482
+ required = _required_fields(raw, _MANIFEST_KEYS - {"derived"}, f"entry {key!r}")
483
+ source_id = _as_str(required["source_id"], "source_id")
484
+ _validate_source_id(key, source_id)
485
+
486
+ derived_raw = raw.get("derived", {})
487
+ if not isinstance(derived_raw, dict):
488
+ raise A2LError(f"derived artifacts for {key!r} must be an object")
489
+ derived: dict[str, DerivedArtifact] = {}
490
+ for name, artifact in derived_raw.items():
491
+ if not isinstance(name, str) or not name or "/" in name or "\\" in name:
492
+ raise A2LError("derived artifact names must be simple non-empty strings")
493
+ derived[name] = _artifact_from_json(name, artifact, required["sha256"])
494
+
495
+ return _validate_entry(
496
+ ManifestEntry(
497
+ path=_as_str(required["path"], "path"),
498
+ sha256=_as_str(required["sha256"], "sha256"),
499
+ source_id=source_id,
500
+ etag=_as_optional_str(required["etag"], "etag"),
501
+ last_modified=_as_optional_str(required["last_modified"], "last_modified"),
502
+ size=_as_nonnegative_int(required["size"], "size"),
503
+ fetched_at=_as_str(required["fetched_at"], "fetched_at"),
504
+ derived=derived,
505
+ )
506
+ )
507
+
508
+
509
+ def _artifact_from_json(name: str, raw: object, source_sha256: object) -> DerivedArtifact:
510
+ if not isinstance(raw, dict):
511
+ raise A2LError(f"derived artifact {name!r} must be an object")
512
+ unknown = set(raw) - _DERIVED_KEYS
513
+ if unknown:
514
+ raise A2LError(f"derived artifact {name!r} has unknown fields: {sorted(unknown)!r}")
515
+ required = _required_fields(raw, _DERIVED_REQUIRED_KEYS, f"derived artifact {name!r}")
516
+ if required["source_sha256"] != source_sha256:
517
+ raise A2LError(f"derived artifact {name!r} source_sha256 does not match parent")
518
+ return _validate_artifact(
519
+ DerivedArtifact(
520
+ path=_as_str(required["path"], "derived path"),
521
+ sha256=_as_str(required["sha256"], "derived sha256"),
522
+ source_sha256=_as_str(required["source_sha256"], "source_sha256"),
523
+ tool=_as_str(required["tool"], "tool"),
524
+ tool_version=_as_str(required["tool_version"], "tool_version"),
525
+ created_at=_as_str(required["created_at"], "created_at"),
526
+ ocr_words_per_page=_as_optional_positive_int(
527
+ raw.get("ocr_words_per_page"), "ocr_words_per_page"
528
+ ),
529
+ page_coverage=_as_page_coverage(raw.get("page_coverage", ())),
530
+ )
531
+ )
532
+
533
+
534
+ def _validate_entry(entry: ManifestEntry) -> ManifestEntry:
535
+ if not isinstance(entry, ManifestEntry):
536
+ raise A2LError("manifest entries must be ManifestEntry values")
537
+ _validate_relative_posix(entry.path, field="relative POSIX path")
538
+ _validate_hash(entry.sha256, "sha256")
539
+ if not isinstance(entry.source_id, str) or not entry.source_id:
540
+ raise A2LError("source_id must be a non-empty string")
541
+ if entry.etag is not None and not isinstance(entry.etag, str):
542
+ raise A2LError("etag must be a string or null")
543
+ if entry.last_modified is not None and not isinstance(entry.last_modified, str):
544
+ raise A2LError("last_modified must be a string or null")
545
+ if isinstance(entry.size, bool) or not isinstance(entry.size, int) or entry.size < 0:
546
+ raise A2LError("size must be a non-negative integer")
547
+ fetched_at = _manifest_timestamp(entry.fetched_at, "fetched_at")
548
+ derived: dict[str, DerivedArtifact] = {}
549
+ if not isinstance(entry.derived, Mapping):
550
+ raise A2LError("derived artifacts must be a mapping")
551
+ derived_paths: set[str] = set()
552
+ for name, artifact in entry.derived.items():
553
+ if not isinstance(name, str) or not name or "/" in name or "\\" in name:
554
+ raise A2LError("derived artifact names must be simple non-empty strings")
555
+ checked = _validate_artifact(artifact)
556
+ if checked.source_sha256 != entry.sha256:
557
+ raise A2LError(f"derived artifact {name!r} source_sha256 does not match parent")
558
+ if checked.path == entry.path:
559
+ raise A2LError(f"derived artifact {name!r} path must differ from source path")
560
+ if checked.path in derived_paths:
561
+ raise A2LError(f"derived artifact {name!r} duplicates another artifact path")
562
+ derived_paths.add(checked.path)
563
+ derived[name] = checked
564
+ return ManifestEntry(
565
+ path=entry.path,
566
+ sha256=entry.sha256,
567
+ source_id=entry.source_id,
568
+ etag=entry.etag,
569
+ last_modified=entry.last_modified,
570
+ size=entry.size,
571
+ fetched_at=fetched_at,
572
+ derived=derived,
573
+ )
574
+
575
+
576
+ def _validate_source_entry(key: str, entry: ManifestEntry) -> ManifestEntry:
577
+ _validate_source_key(key)
578
+ checked = _validate_entry(entry)
579
+ _validate_source_id(key, checked.source_id)
580
+ return checked
581
+
582
+
583
+ def _validate_source_id(key: str, source_id: str) -> None:
584
+ if source_id != key.rsplit(":", 1)[1]:
585
+ raise A2LError(f"source_id for {key!r} does not match its canonical entity ID")
586
+
587
+
588
+ def _validate_artifact(artifact: DerivedArtifact) -> DerivedArtifact:
589
+ if not isinstance(artifact, DerivedArtifact):
590
+ raise A2LError("derived artifacts must be DerivedArtifact values")
591
+ _validate_relative_posix(artifact.path, field="derived relative POSIX path")
592
+ _validate_hash(artifact.sha256, "derived sha256")
593
+ _validate_hash(artifact.source_sha256, "source_sha256")
594
+ for field_name in ("tool", "tool_version"):
595
+ value = getattr(artifact, field_name)
596
+ if not isinstance(value, str) or not value:
597
+ raise A2LError(f"{field_name} must be a non-empty string")
598
+ ocr_words_per_page = _as_optional_positive_int(
599
+ artifact.ocr_words_per_page, "ocr_words_per_page"
600
+ )
601
+ page_coverage = _as_page_coverage(artifact.page_coverage)
602
+ return DerivedArtifact(
603
+ path=artifact.path,
604
+ sha256=artifact.sha256,
605
+ source_sha256=artifact.source_sha256,
606
+ tool=artifact.tool,
607
+ tool_version=artifact.tool_version,
608
+ created_at=_manifest_timestamp(artifact.created_at, "created_at"),
609
+ ocr_words_per_page=ocr_words_per_page,
610
+ page_coverage=page_coverage,
611
+ )
612
+
613
+
614
+ def _entry_to_json(entry: ManifestEntry) -> dict[str, object]:
615
+ checked = _validate_entry(entry)
616
+ return {
617
+ "derived": {
618
+ name: {
619
+ "created_at": artifact.created_at,
620
+ "path": artifact.path,
621
+ "sha256": artifact.sha256,
622
+ "source_sha256": artifact.source_sha256,
623
+ "tool": artifact.tool,
624
+ "tool_version": artifact.tool_version,
625
+ **(
626
+ {"ocr_words_per_page": artifact.ocr_words_per_page}
627
+ if artifact.ocr_words_per_page is not None
628
+ else {}
629
+ ),
630
+ **(
631
+ {"page_coverage": [dict(page) for page in artifact.page_coverage]}
632
+ if artifact.page_coverage
633
+ else {}
634
+ ),
635
+ }
636
+ for name, artifact in sorted(checked.derived.items(), key=lambda item: item[0])
637
+ },
638
+ "etag": checked.etag,
639
+ "fetched_at": checked.fetched_at,
640
+ "last_modified": checked.last_modified,
641
+ "path": checked.path,
642
+ "sha256": checked.sha256,
643
+ "size": checked.size,
644
+ "source_id": checked.source_id,
645
+ }
646
+
647
+
648
+ def _validate_source_key(key: str) -> None:
649
+ if not isinstance(key, str):
650
+ raise A2LError("canonical source key must be a string")
651
+ parts = key.split(":")
652
+ if len(parts) != 4 or any(not _KEY_PART.fullmatch(part) for part in parts):
653
+ raise A2LError("canonical source key must be school:course_org_unit:entity_kind:entity_id")
654
+
655
+
656
+ def _validate_relative_posix(value: object, *, field: str) -> None:
657
+ if not isinstance(value, str) or not value:
658
+ raise A2LError(f"{field} must be a non-empty relative POSIX path")
659
+ if "\\" in value or value.startswith("/") or _WINDOWS_ABSOLUTE.match(value):
660
+ raise A2LError(f"{field} must be relative POSIX")
661
+ pure = PurePosixPath(value)
662
+ if pure.is_absolute() or any(part in {"", ".", ".."} for part in pure.parts):
663
+ raise A2LError(f"{field} must be relative POSIX and cannot escape the root")
664
+ if pure.as_posix() != value:
665
+ raise A2LError(f"{field} must be normalized relative POSIX")
666
+
667
+
668
+ def _validate_hash(value: object, field: str) -> None:
669
+ if not isinstance(value, str) or _SHA256.fullmatch(value) is None:
670
+ raise A2LError(f"{field} must be a lowercase 64-character SHA-256")
671
+
672
+
673
+ def _manifest_timestamp(value: object, field: str) -> str:
674
+ if not isinstance(value, str):
675
+ raise A2LError(f"{field} must be a timezone-aware ISO 8601 UTC timestamp")
676
+ parsed = _parse_timestamp(value, field)
677
+ if parsed.utcoffset() != timedelta(0):
678
+ raise A2LError(f"{field} must be a timezone-aware ISO 8601 UTC timestamp")
679
+ return parsed.astimezone(UTC).isoformat().replace("+00:00", "Z")
680
+
681
+
682
+ def _parse_timestamp(value: str, field: str) -> datetime:
683
+ try:
684
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
685
+ except ValueError as exc:
686
+ raise A2LError(f"{field} must be a timezone-aware ISO 8601 UTC timestamp") from exc
687
+ if parsed.tzinfo is None or parsed.utcoffset() is None:
688
+ raise A2LError(f"{field} must be timezone-aware")
689
+ return parsed
690
+
691
+
692
+ def _utc_datetime(value: datetime, field: str) -> datetime:
693
+ if not isinstance(value, datetime) or value.tzinfo is None or value.utcoffset() is None:
694
+ raise A2LError(f"{field} must be timezone-aware UTC")
695
+ return value.astimezone(UTC)
696
+
697
+
698
+ def _required_fields(
699
+ raw: Mapping[str, object], fields: Collection[str], context: str
700
+ ) -> dict[str, object]:
701
+ missing = set(fields) - raw.keys()
702
+ if missing:
703
+ raise A2LError(f"{context} is missing fields: {sorted(missing)!r}")
704
+ return {key: raw[key] for key in fields}
705
+
706
+
707
+ def _as_str(value: object, field: str) -> str:
708
+ if not isinstance(value, str):
709
+ raise A2LError(f"{field} must be a string")
710
+ return value
711
+
712
+
713
+ def _as_optional_str(value: object, field: str) -> str | None:
714
+ if value is None:
715
+ return None
716
+ if not isinstance(value, str):
717
+ raise A2LError(f"{field} must be a string or null")
718
+ return value
719
+
720
+
721
+ def _as_nonnegative_int(value: object, field: str) -> int:
722
+ if isinstance(value, bool) or not isinstance(value, int) or value < 0:
723
+ raise A2LError(f"{field} must be a non-negative integer")
724
+ return value
725
+
726
+
727
+ def _as_optional_positive_int(value: object, field: str) -> int | None:
728
+ if value is None:
729
+ return None
730
+ if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
731
+ raise A2LError(f"{field} must be a positive integer or null")
732
+ return value
733
+
734
+
735
+ def _as_page_coverage(value: object) -> tuple[Mapping[str, object], ...]:
736
+ if value is None:
737
+ return ()
738
+ if not isinstance(value, (list, tuple)):
739
+ raise A2LError("page_coverage must be an array")
740
+ normalized: list[Mapping[str, object]] = []
741
+ for index, raw_page in enumerate(value, start=1):
742
+ if not isinstance(raw_page, Mapping):
743
+ raise A2LError(f"page_coverage item {index} must be an object")
744
+ page = raw_page.get("page")
745
+ mode = raw_page.get("mode")
746
+ words = raw_page.get("words")
747
+ warning = raw_page.get("warning")
748
+ if isinstance(page, bool) or not isinstance(page, int) or page <= 0:
749
+ raise A2LError(f"page_coverage item {index} page must be a positive integer")
750
+ if not isinstance(mode, str) or not mode:
751
+ raise A2LError(f"page_coverage item {index} mode must be a non-empty string")
752
+ if isinstance(words, bool) or not isinstance(words, int) or words < 0:
753
+ raise A2LError(f"page_coverage item {index} words must be a non-negative integer")
754
+ if warning is not None and not isinstance(warning, str):
755
+ raise A2LError(f"page_coverage item {index} warning must be a string or null")
756
+ normalized.append({"page": page, "mode": mode, "words": words, "warning": warning})
757
+ return tuple(normalized)
758
+
759
+
760
+ def _read_json_object(destination: Path, label: str) -> dict[str, object] | None:
761
+ try:
762
+ with open(os.fspath(paths.long_path(destination)), encoding="utf-8", newline="") as handle:
763
+ raw: Any = json.load(handle)
764
+ except FileNotFoundError:
765
+ return None
766
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
767
+ message = (
768
+ f"{label} is unreadable" if isinstance(exc, OSError) else f"{label} is not valid JSON"
769
+ )
770
+ raise A2LError(message) from exc
771
+ if not isinstance(raw, dict):
772
+ raise A2LError(f"{label} root must be an object")
773
+ return cast(dict[str, object], raw)
774
+
775
+
776
+ def _read_text(destination: Path) -> str:
777
+ try:
778
+ with open(os.fspath(paths.long_path(destination)), encoding="utf-8", newline="") as handle:
779
+ return handle.read()
780
+ except (OSError, UnicodeError) as exc:
781
+ raise A2LError("vault VERSION is unreadable") from exc
782
+
783
+
784
+ def _regular_file_exists(path: Path, label: str) -> bool:
785
+ try:
786
+ file_stat = os.lstat(os.fspath(paths.long_path(path)))
787
+ except FileNotFoundError:
788
+ return False
789
+ except OSError as exc:
790
+ raise A2LError(f"{label} is unreadable") from exc
791
+ if paths.is_link(path) or stat.S_ISLNK(file_stat.st_mode):
792
+ raise A2LError(f"{label} must not be a symlink")
793
+ if not stat.S_ISREG(file_stat.st_mode):
794
+ raise A2LError(f"{label} must be a regular file")
795
+ return True
796
+
797
+
798
+ def _canonical_json(payload: object) -> str:
799
+ return (
800
+ json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2, separators=(",", ": "))
801
+ + "\n"
802
+ )
803
+
804
+
805
+ def _hash_file(source: Path) -> tuple[str, int]:
806
+ digest = sha256()
807
+ size = 0
808
+ try:
809
+ with open(os.fspath(paths.long_path(source)), "rb") as handle:
810
+ while chunk := handle.read(_COPY_CHUNK_SIZE):
811
+ digest.update(chunk)
812
+ size += len(chunk)
813
+ except (FileNotFoundError, IsADirectoryError):
814
+ return "", -1
815
+ return digest.hexdigest(), size
816
+
817
+
818
+ def _copy_verified(
819
+ source: Path,
820
+ destination: Path,
821
+ *,
822
+ root: Path | None = None,
823
+ expected_sha256: str | None,
824
+ expected_size: int | None,
825
+ ) -> bool:
826
+ """Stream a source into a sibling ``.part`` and atomically install it if valid."""
827
+
828
+ paths.ensure_dir(destination.parent, root=root)
829
+ if root is not None and paths.has_link_component(destination.parent, root=root):
830
+ raise A2LError("history destination parent contains a link component")
831
+ file_descriptor, raw_temporary = tempfile.mkstemp(
832
+ prefix=f".{destination.name}.",
833
+ suffix=".part",
834
+ dir=os.fspath(paths.long_path(destination.parent)),
835
+ )
836
+ temporary = paths.plain_path(Path(raw_temporary))
837
+ digest = sha256()
838
+ size = 0
839
+ descriptor_open = True
840
+ try:
841
+ # The temporary reservation happens after the destination-parent check. Revalidate the
842
+ # reserved path before opening it so a swapped parent cannot redirect a history copy.
843
+ if root is not None and paths.has_link_component(temporary, root=root):
844
+ raise A2LError("history temporary path contains a link component")
845
+ with (
846
+ open(os.fspath(paths.long_path(source)), "rb") as input_handle,
847
+ os.fdopen(file_descriptor, "wb") as output_handle,
848
+ ):
849
+ descriptor_open = False
850
+ while chunk := input_handle.read(_COPY_CHUNK_SIZE):
851
+ digest.update(chunk)
852
+ size += len(chunk)
853
+ output_handle.write(chunk)
854
+ output_handle.flush()
855
+ os.fsync(output_handle.fileno())
856
+ actual_sha256 = digest.hexdigest()
857
+ if expected_sha256 is not None and (
858
+ actual_sha256 != expected_sha256
859
+ or (expected_size is not None and size != expected_size)
860
+ ):
861
+ raise A2LError("manifest source hash mismatch: integrity gap")
862
+ paths.atomic_install_temp(destination, temporary, root=root)
863
+ return True
864
+ except (FileNotFoundError, IsADirectoryError):
865
+ return False
866
+ finally:
867
+ # This temporary is a generated local backup copy and is cheap to recreate; unlike an
868
+ # ingest download .part, cleanup after an install failure is intentional here.
869
+ if descriptor_open:
870
+ with suppress(OSError):
871
+ os.close(file_descriptor)
872
+ with suppress(OSError):
873
+ os.unlink(os.fspath(paths.long_path(temporary)))
874
+
875
+
876
+ def _allocate_revision_directory(bucket: Path, timestamp: str, *, root: Path | None = None) -> Path:
877
+ for number in range(1, 100_000):
878
+ suffix = "" if number == 1 else f"_{number}"
879
+ candidate = bucket / f"{timestamp}{suffix}"
880
+ if root is not None and paths.has_link_component(candidate, root=root):
881
+ raise A2LError("revision path contains a link component")
882
+ try:
883
+ paths.long_path(candidate).mkdir(parents=False, exist_ok=False)
884
+ except FileExistsError:
885
+ continue
886
+ if root is not None and paths.has_link_component(candidate, root=root):
887
+ raise A2LError("revision path contains a link component")
888
+ return candidate
889
+ raise A2LError("could not allocate a collision-safe revision directory")
890
+
891
+
892
+ def _copy_tree(source: Path, destination: Path, *, root: Path | None = None) -> None:
893
+ """Copy a tree while applying the long-path boundary to each individual operation."""
894
+ if paths.is_link(source):
895
+ raise A2LError("schema state cannot contain symlinks")
896
+ if paths.is_link(destination):
897
+ raise A2LError("schema staging destination cannot be a symlink")
898
+ paths.ensure_dir(destination, root=root)
899
+ for candidate in paths.walk(source):
900
+ relative = candidate.relative_to(source)
901
+ target = destination / relative
902
+ if paths.is_link(candidate):
903
+ raise A2LError("schema state cannot contain symlinks")
904
+ if paths.long_path(candidate).is_dir():
905
+ paths.ensure_dir(target, root=root)
906
+ continue
907
+ if not paths.long_path(candidate).is_file():
908
+ raise A2LError("schema state contains an unsupported filesystem entry")
909
+ paths.ensure_dir(target.parent, root=root)
910
+ with (
911
+ open(os.fspath(paths.long_path(candidate)), "rb") as source_handle,
912
+ _open_private_copy_target(target, root=root) as destination_handle,
913
+ ):
914
+ while chunk := source_handle.read(_COPY_CHUNK_SIZE):
915
+ destination_handle.write(chunk)
916
+
917
+
918
+ def _open_private_copy_target(path: Path, *, root: Path | None = None) -> BinaryIO:
919
+ """Create a new state-copy file without following a replacement link."""
920
+
921
+ if root is not None and paths.has_link_component(path, root=root):
922
+ raise A2LError("schema state copy path contains a link component")
923
+ flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0)
924
+ file_descriptor = os.open(os.fspath(paths.long_path(path)), flags, 0o600)
925
+ try:
926
+ file_stat = os.fstat(file_descriptor)
927
+ if not stat.S_ISREG(file_stat.st_mode) or getattr(file_stat, "st_nlink", 1) != 1:
928
+ raise A2LError("schema state copy must be a private regular file")
929
+ handle = os.fdopen(file_descriptor, "wb")
930
+ file_descriptor = -1
931
+ return handle
932
+ except BaseException:
933
+ if file_descriptor >= 0:
934
+ os.close(file_descriptor)
935
+ raise
936
+
937
+
938
+ def _backup_state(state: Path, version: int, *, root: Path | None = None) -> Path:
939
+ base = state.parent / f".a2l-backup-v{version}"
940
+ candidate = base
941
+ number = 2
942
+ while _occupied(candidate):
943
+ candidate = state.parent / f".a2l-backup-v{version}-{number}"
944
+ number += 1
945
+ try:
946
+ _copy_tree(state, candidate, root=root)
947
+ except BaseException:
948
+ paths.remove_tree(candidate, ignore_errors=True)
949
+ raise
950
+ return candidate
951
+
952
+
953
+ def _install_migrated_state(
954
+ state: Path, staged_state: Path, backup: Path, *, root: Path | None = None
955
+ ) -> None:
956
+ """Commit staged schema state, restoring the backup if installation fails."""
957
+
958
+ try:
959
+ _synchronize_state(staged_state, state, root=root)
960
+ except BaseException:
961
+ try:
962
+ _synchronize_state(backup, state, root=root)
963
+ except BaseException as restore_error:
964
+ raise A2LError("schema migration failed and rollback also failed") from restore_error
965
+ raise
966
+
967
+
968
+ def _synchronize_state(source: Path, destination: Path, *, root: Path | None = None) -> None:
969
+ """Synchronize one state tree into another with VERSION installed last."""
970
+
971
+ source_files, source_directories = _state_tree(source)
972
+ destination_files, destination_directories = _state_tree(destination)
973
+
974
+ obsolete = (destination_files - source_files) | (destination_directories - source_directories)
975
+ for relative in sorted(obsolete, key=_deepest_first):
976
+ _remove_state_path(destination / _relative_path(relative), root=root)
977
+
978
+ for relative in sorted(source_directories, key=_shallowest_first):
979
+ directory = destination / _relative_path(relative)
980
+ paths.ensure_dir(directory, root=root)
981
+
982
+ for relative in sorted(source_files - {"VERSION"}):
983
+ source_file = source / _relative_path(relative)
984
+ destination_file = destination / _relative_path(relative)
985
+ if not _same_file(source_file, destination_file):
986
+ paths.atomic_write_bytes(destination_file, _read_bytes(source_file), root=root)
987
+
988
+ version_source = source / "VERSION"
989
+ if "VERSION" not in source_files:
990
+ raise A2LError("staged schema state is missing VERSION")
991
+ paths.atomic_write_bytes(destination / "VERSION", _read_bytes(version_source), root=root)
992
+
993
+
994
+ def _state_tree(root: Path) -> tuple[set[str], set[str]]:
995
+ files: set[str] = set()
996
+ directories: set[str] = set()
997
+ for candidate in paths.walk(root):
998
+ if paths.is_link(candidate):
999
+ raise A2LError("schema state cannot contain symlinks")
1000
+ relative = candidate.relative_to(root).as_posix()
1001
+ if paths.long_path(candidate).is_file():
1002
+ files.add(relative)
1003
+ elif paths.long_path(candidate).is_dir():
1004
+ directories.add(relative)
1005
+ else:
1006
+ raise A2LError("schema state contains an unsupported filesystem entry")
1007
+ return files, directories
1008
+
1009
+
1010
+ def _relative_path(relative: str) -> Path:
1011
+ return Path(*PurePosixPath(relative).parts)
1012
+
1013
+
1014
+ def _deepest_first(relative: str) -> tuple[int, str]:
1015
+ return (-len(PurePosixPath(relative).parts), relative)
1016
+
1017
+
1018
+ def _shallowest_first(relative: str) -> tuple[int, str]:
1019
+ return (len(PurePosixPath(relative).parts), relative)
1020
+
1021
+
1022
+ def _remove_state_path(path: Path, *, root: Path | None = None) -> None:
1023
+ if root is not None and paths.has_link_component(path, root=root):
1024
+ raise A2LError("schema state path contains a link component")
1025
+ if paths.long_path(path).is_dir() and not paths.is_link(path):
1026
+ paths.remove_tree(path)
1027
+ else:
1028
+ os.unlink(os.fspath(paths.long_path(path)))
1029
+
1030
+
1031
+ def _same_file(first: Path, second: Path) -> bool:
1032
+ if not paths.long_path(first).is_file() or not paths.long_path(second).is_file():
1033
+ return False
1034
+ return _hash_file(first) == _hash_file(second)
1035
+
1036
+
1037
+ def _read_bytes(source: Path) -> bytes:
1038
+ with open(os.fspath(paths.long_path(source)), "rb") as handle:
1039
+ return handle.read()
1040
+
1041
+
1042
+ def _occupied(path: Path) -> bool:
1043
+ return paths.long_path(path).exists() or paths.is_link(path)
1044
+
1045
+
1046
+ def _write_vault_gitignore(root: Path) -> None:
1047
+ paths.atomic_write_text(
1048
+ root / ".gitignore",
1049
+ ".a2l/\n**/_meta/my_grades.json\n**/discussions/\n**/.a2l/submissions/\n",
1050
+ root=root,
1051
+ )
1052
+
1053
+
1054
+ def _refuse_agent2learn_checkout(path: Path) -> None:
1055
+ source_root = _agent2learn_source_root()
1056
+ if source_root is None:
1057
+ return
1058
+ try:
1059
+ path.relative_to(source_root)
1060
+ except ValueError:
1061
+ return
1062
+ raise A2LError("refusing to place a vault inside the Agent2Learn source checkout")
1063
+
1064
+
1065
+ def _require_git_confirmation_if_needed(path: Path) -> None:
1066
+ repository_root = _git_root(path)
1067
+ if repository_root is None:
1068
+ return
1069
+ if not sys.stdin.isatty():
1070
+ raise A2LError(
1071
+ f"selected vault is inside Git worktree {repository_root}; explicit TTY confirmation "
1072
+ "is required"
1073
+ )
1074
+ try:
1075
+ answer = input(
1076
+ f"Selected vault is inside Git worktree {repository_root}. "
1077
+ f"Type {_GIT_CONFIRMATION!r} to continue: "
1078
+ )
1079
+ except EOFError as exc:
1080
+ raise A2LError("vault claim cancelled: no TTY confirmation") from exc
1081
+ if answer.strip() != _GIT_CONFIRMATION:
1082
+ raise A2LError("vault claim cancelled: explicit Git worktree confirmation did not match")
1083
+
1084
+
1085
+ def _agent2learn_source_root() -> Path | None:
1086
+ package_root = Path(__file__).resolve().parents[2]
1087
+ repository_root = _git_root(package_root)
1088
+ if repository_root is None:
1089
+ return None
1090
+ if (
1091
+ paths.long_path(repository_root / "pyproject.toml").is_file()
1092
+ and paths.long_path(repository_root / "src" / "agent2learn").is_dir()
1093
+ ):
1094
+ return repository_root
1095
+ return None
1096
+
1097
+
1098
+ def _git_root(path: Path) -> Path | None:
1099
+ probe = path if paths.long_path(path).is_dir() else path.parent
1100
+ try:
1101
+ result = subprocess.run(
1102
+ [
1103
+ "git",
1104
+ "-C",
1105
+ os.fspath(paths.long_path(probe)),
1106
+ "rev-parse",
1107
+ "--show-toplevel",
1108
+ ],
1109
+ check=False,
1110
+ capture_output=True,
1111
+ text=True,
1112
+ encoding="utf-8",
1113
+ )
1114
+ except (OSError, subprocess.SubprocessError) as exc:
1115
+ raise A2LError("could not establish Git worktree status") from exc
1116
+ if result.returncode != 0 or not result.stdout.strip():
1117
+ stderr = getattr(result, "stderr", "")
1118
+ if isinstance(stderr, str) and "not a git repository" in stderr.casefold():
1119
+ return None
1120
+ raise A2LError("could not establish Git worktree status")
1121
+ try:
1122
+ return Path(result.stdout.strip()).resolve()
1123
+ except (OSError, RuntimeError, ValueError) as exc:
1124
+ raise A2LError("could not establish Git worktree status") from exc
1125
+
1126
+
1127
+ __all__ = [
1128
+ "MIGRATIONS",
1129
+ "SCHEMA_VERSION",
1130
+ "DerivedArtifact",
1131
+ "ManifestEntry",
1132
+ "Vault",
1133
+ "check_schema",
1134
+ ]