agent2learn 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. agent2learn/__init__.py +3 -0
  2. agent2learn/_release.py +19 -0
  3. agent2learn/aipolicy.py +182 -0
  4. agent2learn/api.py +590 -0
  5. agent2learn/audit.py +358 -0
  6. agent2learn/auth/__init__.py +282 -0
  7. agent2learn/auth/cdp.py +1067 -0
  8. agent2learn/auth/paste.py +378 -0
  9. agent2learn/calendar.py +525 -0
  10. agent2learn/calibrate.py +347 -0
  11. agent2learn/check.py +1091 -0
  12. agent2learn/cli.py +2039 -0
  13. agent2learn/clock.py +39 -0
  14. agent2learn/config.py +205 -0
  15. agent2learn/console.py +229 -0
  16. agent2learn/convert.py +1223 -0
  17. agent2learn/doctor.py +1167 -0
  18. agent2learn/errors.py +32 -0
  19. agent2learn/ground.py +735 -0
  20. agent2learn/index.py +614 -0
  21. agent2learn/ingest.py +3229 -0
  22. agent2learn/locations.py +247 -0
  23. agent2learn/outlines.py +754 -0
  24. agent2learn/paths.py +683 -0
  25. agent2learn/pipeline.py +392 -0
  26. agent2learn/privacy.py +1123 -0
  27. agent2learn/schools/__init__.py +29 -0
  28. agent2learn/schools/_base.py +194 -0
  29. agent2learn/schools/generic.py +78 -0
  30. agent2learn/schools/uwaterloo.py +66 -0
  31. agent2learn/session.py +373 -0
  32. agent2learn/skills.py +1081 -0
  33. agent2learn/snapshot.py +399 -0
  34. agent2learn/submit.py +1047 -0
  35. agent2learn/transactions.py +157 -0
  36. agent2learn/upgrade.py +288 -0
  37. agent2learn/vault.py +1134 -0
  38. agent2learn-0.1.2.data/data/a2l-coursework/SKILL.md +52 -0
  39. agent2learn-0.1.2.data/data/a2l-setup/SKILL.md +27 -0
  40. agent2learn-0.1.2.data/data/a2l-study/SKILL.md +27 -0
  41. agent2learn-0.1.2.data/data/a2l-sync/SKILL.md +30 -0
  42. agent2learn-0.1.2.dist-info/METADATA +186 -0
  43. agent2learn-0.1.2.dist-info/RECORD +46 -0
  44. agent2learn-0.1.2.dist-info/WHEEL +4 -0
  45. agent2learn-0.1.2.dist-info/entry_points.txt +3 -0
  46. agent2learn-0.1.2.dist-info/licenses/LICENSE +202 -0
agent2learn/check.py ADDED
@@ -0,0 +1,1091 @@
1
+ """An experimental lexical evidence scan over the student's own course material.
2
+
3
+ ``a2l check`` answers one narrow question per claim: *what matching or related text did a
4
+ deterministic lexical retrieval find in this course's verified material, and where is it?* It
5
+ does not grade, prove, rewrite, or answer. Every human-readable report opens with the disclosure
6
+ in :data:`DISCLOSURE`, and no status in this module means correct, incorrect, policy-compliant, or
7
+ academically acceptable. The semantic judgement belongs to the person or agent reading the
8
+ citations.
9
+
10
+ Three decisions keep it honest.
11
+
12
+ **It cannot manufacture its own evidence.** The source set comes from ``ground.py``, so a
13
+ candidate must trace to a LEARN source ID with an archived original and a markdown twin that both
14
+ still hash to their recorded digests. A student's own draft, a downloaded solution, an untracked
15
+ sibling, and every Agent2Learn-generated report are therefore unreachable — otherwise a claim
16
+ could cite an answer back to itself.
17
+
18
+ **It is exactly reproducible.** Scores are computed with :class:`fractions.Fraction` and
19
+ serialised as floored integer basis points, so no platform can disagree through floating-point
20
+ rounding or JSON formatting. Candidates are ordered by ``(-score_bp, path, line)``. There are no
21
+ embeddings and no model calls.
22
+
23
+ **``possible_conflict`` is deliberately weak.** It fires only for two allowlisted surface forms
24
+ whose token sequences are otherwise identical: opposite comparison operators, or opposite
25
+ ``is``/``is not`` polarity. A differing number never qualifies, because the lecture's ``n = 10``
26
+ and the lab's ``n = 20`` are usually both correct. A false contradiction is worse than no tool at
27
+ all: it would lead a student to "fix" a right answer using the authority of their own course
28
+ notes. The status is rendered as an invitation to compare, never as a claim that the student is
29
+ wrong.
30
+
31
+ ``CANDIDATE_FLOOR_BP``, ``STRONG_MATCH_FLOOR_BP``, the tokeniser, the ``GENERIC`` stopwords, and
32
+ the segmentation heuristic together define :data:`CHECK_ALGORITHM_VERSION`. Changing any of them
33
+ requires a version bump and fixture review.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import heapq
39
+ import json
40
+ import os
41
+ import re
42
+ from collections import Counter
43
+ from collections.abc import Iterator, Mapping, Sequence
44
+ from dataclasses import dataclass, field
45
+ from difflib import SequenceMatcher
46
+ from fractions import Fraction
47
+ from pathlib import Path, PurePosixPath
48
+ from typing import Literal
49
+
50
+ from agent2learn import ground, paths
51
+ from agent2learn import index as course_index
52
+ from agent2learn.errors import A2LError
53
+ from agent2learn.ground import GENERIC, tok
54
+ from agent2learn.vault import Vault
55
+
56
+ CHECK_ALGORITHM_VERSION = 1
57
+ CANDIDATE_FLOOR_BP = 3_500
58
+ STRONG_MATCH_FLOOR_BP = 7_500
59
+ NOTATION_FLOOR_BP = 7_200
60
+ TOP_CITATIONS = 5
61
+ DISCLOSURE = "Experimental lexical evidence scan — review the cited sources yourself."
62
+
63
+ SUPPORTED_SUFFIXES = frozenset({".md", ".txt", ".ipynb", ".py", ".r", ".rmd", ".tex"})
64
+
65
+ ClaimKind = Literal["prose", "code", "formula", "step"]
66
+ Status = Literal[
67
+ "evidence_found",
68
+ "related_evidence",
69
+ "no_matching_evidence",
70
+ "possible_conflict",
71
+ "skipped",
72
+ ]
73
+
74
+ _FETCHABLE = frozenset({"metadata_only", "source_only", "integrity_gap", "download_gap"})
75
+ _AVAILABILITY_NOTES = {
76
+ "metadata_only": "known from metadata but never fetched",
77
+ "source_only": "fetched, but it has no markdown twin yet",
78
+ "integrity_gap": "on-disk bytes no longer match the manifest",
79
+ "download_gap": "the server did not serve this file; retry the fetch",
80
+ "conversion_gap": "fetched, but conversion produced no markdown twin",
81
+ "unsupported_format": "no converter handles this format",
82
+ "external_link": "external or licensed link, deliberately not fetched",
83
+ }
84
+
85
+ # Function and discourse words. Distinct from GENERIC, which removes coursework-title noise from
86
+ # retrieval scoring: this set decides whether a sentence carries any checkable content at all.
87
+ _FUNCTION = frozenset(
88
+ {
89
+ "a", "about", "above", "after", "again", "all", "also", "although", "am", "an", "and",
90
+ "another", "any", "are", "as", "at", "be", "because", "been", "before", "being", "below",
91
+ "both", "but", "by", "can", "could", "did", "do", "does", "during", "each", "else",
92
+ "finally", "first", "following", "for", "from", "further", "had", "has", "have", "he",
93
+ "hence", "her", "here", "him", "his", "how", "however", "i", "if", "in", "into", "is",
94
+ "it", "its", "just", "least", "less", "many", "may", "me", "might", "more", "most",
95
+ "must", "my", "never", "next", "no", "not", "now", "of", "on", "one", "only", "or",
96
+ "other", "our", "over", "per", "quite", "rather", "same", "second", "shall", "she",
97
+ "should", "since", "so", "some", "still", "such", "than", "that", "the", "their", "them",
98
+ "then", "there", "therefore", "these", "they", "third", "this", "those", "though",
99
+ "thus", "to", "under", "until", "up", "us", "very", "via", "was", "we", "were", "what",
100
+ "when", "where", "which", "while", "who", "whom", "whose", "why", "will", "with",
101
+ "within", "without", "would", "yet", "you", "your",
102
+ }
103
+ ) # fmt: skip
104
+
105
+ _SENTENCE_SPLIT = re.compile(r"(?<=[.!?])\s+")
106
+ # A markdown image is not a statement. Stripped before segmentation so a data-URI or file path
107
+ # cannot become junk tokens, inflate a claim's term count, or "match" an identical blob upstream.
108
+ _MARKDOWN_IMAGE = re.compile(r"!\[[^\]]*\]\([^)]*\)")
109
+ _STEP = re.compile(r"^\s*\d+\s*[.)]\s+")
110
+ _NUMBER_OR_MATH = re.compile(r"\d|[=<>≤≥≠∈∉∑∏∫√±×÷^{}]")
111
+ _DEFINITION_CUE = re.compile(
112
+ r"\b(?:is|are)\s+defined\b|\bwe\s+(?:define|denote|let)\b|\bdenotes?\b|\bmeans\b|"
113
+ r"\brefers?\s+to\b|\blet\s+\w+\s+be\b",
114
+ re.IGNORECASE,
115
+ )
116
+ _IDENTIFIER = re.compile(
117
+ r"`[^`]+`|[A-Za-z][A-Za-z0-9]*(?:_[A-Za-z0-9]+)+|"
118
+ r"[A-Za-z][A-Za-z0-9]*(?:\.[A-Za-z][A-Za-z0-9]*)+|[a-z]+[A-Z][A-Za-z0-9]*"
119
+ )
120
+ _NAMED_METHOD = re.compile(
121
+ r"\b[A-Z][a-z]+(?:['\u2019]s)?\s+(?:algorithm|method|decomposition|theorem|lemma|rule|"
122
+ r"criterion|transform|test|inequality|bound|relaxation)\b"
123
+ )
124
+ _NUMBER = re.compile(r"\d+(?:\.\d+)?")
125
+ _NEGATION = re.compile(r"\b(?:not|no|never|cannot|without)\b|n['\u2019]t\b", re.IGNORECASE)
126
+ _IS_POLARITY = re.compile(r"\bis(?P<not>\s+not)?\b", re.IGNORECASE)
127
+ _RUN = re.compile(r"[a-z0-9]+")
128
+
129
+ # Longest first: "greater than or equal to" must not be consumed as "greater than".
130
+ _WORD_OPERATORS: tuple[tuple[re.Pattern[str], str], ...] = tuple(
131
+ (re.compile(pattern, re.IGNORECASE), replacement)
132
+ for pattern, replacement in (
133
+ (r"\bgreater\s+than\s+or\s+equal\s+to\b", ">="),
134
+ (r"\bless\s+than\s+or\s+equal\s+to\b", "<="),
135
+ (r"\bno\s+more\s+than\b", "<="),
136
+ (r"\bno\s+less\s+than\b", ">="),
137
+ (r"\bat\s+most\b", "<="),
138
+ (r"\bat\s+least\b", ">="),
139
+ (r"\bstrictly\s+greater\s+than\b", ">"),
140
+ (r"\bstrictly\s+less\s+than\b", "<"),
141
+ (r"\bgreater\s+than\b", ">"),
142
+ (r"\bless\s+than\b", "<"),
143
+ )
144
+ )
145
+ _SYMBOL_OPERATORS = ("<=", ">=", "!=", "==", "≤", "≥", "≠", "∈", "<", ">", "=")
146
+ _CANONICAL_OPERATOR = {"≤": "<=", "≥": ">=", "≠": "!=", "==": "="}
147
+ _OPPOSITES = {
148
+ "<": frozenset({">", ">="}),
149
+ ">": frozenset({"<", "<="}),
150
+ "<=": frozenset({">", ">="}),
151
+ ">=": frozenset({"<", "<="}),
152
+ "=": frozenset({"!="}),
153
+ "!=": frozenset({"="}),
154
+ }
155
+
156
+
157
+ @dataclass(frozen=True)
158
+ class Claim:
159
+ """One checkable unit of the draft, located by its starting line."""
160
+
161
+ line: int
162
+ text: str
163
+ kind: ClaimKind
164
+
165
+
166
+ @dataclass(frozen=True)
167
+ class Citation:
168
+ """One retrieved course span, pinned to the revision that was scanned."""
169
+
170
+ path: str
171
+ line: int
172
+ excerpt: str
173
+ source_sha256: str
174
+ derived_sha256: str
175
+ retrieval_score_bp: int
176
+
177
+
178
+ @dataclass(frozen=True)
179
+ class Finding:
180
+ """What retrieval found for one claim, and nothing more."""
181
+
182
+ claim: Claim
183
+ status: Status
184
+ citations: list[Citation] = field(default_factory=list)
185
+ note: str | None = None
186
+
187
+
188
+ @dataclass(frozen=True)
189
+ class CoverageGap:
190
+ """Course material that could not be scanned, with its honest availability state."""
191
+
192
+ source_key: str
193
+ source_id: str
194
+ title: str
195
+ availability: str
196
+ note: str
197
+ fetch_command: str | None
198
+
199
+
200
+ @dataclass(frozen=True)
201
+ class NotationCandidate:
202
+ """A draft term absent from the scanned material, with a candidate — never a correction."""
203
+
204
+ term: str
205
+ candidate: str | None
206
+ citation: Citation | None
207
+
208
+
209
+ @dataclass(frozen=True)
210
+ class CheckReport:
211
+ """Everything one scan computed, plus the revisions it computed it from."""
212
+
213
+ draft: str
214
+ course: str
215
+ course_code: str
216
+ scope: str
217
+ findings: tuple[Finding, ...]
218
+ coverage_gaps: tuple[CoverageGap, ...]
219
+ notation: tuple[NotationCandidate, ...]
220
+ revisions: dict[str, dict[str, str]]
221
+ algorithm_version: int = CHECK_ALGORITHM_VERSION
222
+
223
+ @property
224
+ def review_required(self) -> bool:
225
+ """Whether ``--strict`` should exit non-zero, as a review reminder only."""
226
+ return any(
227
+ finding.status in {"no_matching_evidence", "possible_conflict"}
228
+ for finding in self.findings
229
+ )
230
+
231
+
232
+ @dataclass(frozen=True)
233
+ class ScanSource:
234
+ """A verified twin on disk, carrying the digests a citation must record."""
235
+
236
+ path: Path
237
+ citation_path: str
238
+ source_sha256: str
239
+ derived_sha256: str
240
+
241
+
242
+ class LineIndex:
243
+ """One in-memory inverted line index per run, so claims do not rescan every file."""
244
+
245
+ def __init__(self, sources: Sequence[ScanSource]) -> None:
246
+ self._lines: list[tuple[ScanSource, int, str, frozenset[str], frozenset[str]]] = []
247
+ self._postings: dict[str, list[int]] = {}
248
+ self._value_postings: dict[str, set[int]] = {}
249
+ self.vocabulary: dict[str, tuple[str, int]] = {}
250
+ for source in sources:
251
+ text = _read_text(source.path)
252
+ if text is None:
253
+ continue
254
+ for number, raw in enumerate(text.splitlines(), start=1):
255
+ terms = frozenset(term for term in tok(raw) if term not in GENERIC)
256
+ if not terms:
257
+ continue
258
+ position = len(self._lines)
259
+ line_values = values(raw)
260
+ self._lines.append((source, number, raw.strip(), terms, line_values))
261
+ for value in line_values:
262
+ self._value_postings.setdefault(value, set()).add(position)
263
+ for term in terms:
264
+ self._postings.setdefault(term, []).append(position)
265
+ self.vocabulary.setdefault(term, (source.citation_path, number))
266
+
267
+ def __len__(self) -> int:
268
+ return len(self._lines)
269
+
270
+ def retrieve(self, claim: Claim, *, top: int = TOP_CITATIONS) -> list[Citation]:
271
+ """Score only the lines that share a term with the claim, then rank deterministically.
272
+
273
+ Term overlap is counted straight off the postings lists rather than by intersecting a set
274
+ per line, and only the surviving spans become :class:`Citation` objects. On a corpus where
275
+ a common term appears on every line, materialising a citation per scored line dominated
276
+ everything else.
277
+ """
278
+
279
+ claim_terms = frozenset(term for term in tok(claim.text) if term not in GENERIC)
280
+ if not claim_terms:
281
+ return []
282
+ claim_values = values(claim.text)
283
+ total_terms = len(claim_terms)
284
+ total_values = len(claim_values)
285
+
286
+ matches: Counter[int] = Counter()
287
+ for term in claim_terms:
288
+ matches.update(self._postings.get(term, ()))
289
+ if not matches:
290
+ return []
291
+ denominator = 5 * total_terms * total_values if total_values else total_terms
292
+ term_scale = 40_000 * total_values if total_values else 10_000
293
+ term_numerators = [matched * term_scale for matched in range(total_terms + 1)]
294
+ value_numerators = [10_000 * total_terms * matched for matched in range(total_values + 1)]
295
+ value_matches: Counter[int] = Counter()
296
+ for value in claim_values:
297
+ positions = self._value_postings.get(value)
298
+ if positions is not None:
299
+ value_matches.update(matches.keys() & positions)
300
+
301
+ def ranked() -> Iterator[tuple[int, str, int, int]]:
302
+ for position, matched in matches.items():
303
+ source, number, _excerpt, _terms, _line_values = self._lines[position]
304
+ matched_values = value_matches.get(position, 0)
305
+ points = (
306
+ term_numerators[matched] + value_numerators[matched_values]
307
+ ) // denominator
308
+ if points > 0:
309
+ yield (-points, source.citation_path, number, position)
310
+
311
+ return [
312
+ self._citation(position, -negated)
313
+ for negated, _path, _line, position in heapq.nsmallest(top, ranked())
314
+ ]
315
+
316
+ def _citation(self, position: int, points: int) -> Citation:
317
+ source, number, excerpt, _terms, _line_values = self._lines[position]
318
+ return Citation(
319
+ path=source.citation_path,
320
+ line=number,
321
+ excerpt=excerpt,
322
+ source_sha256=source.source_sha256,
323
+ derived_sha256=source.derived_sha256,
324
+ retrieval_score_bp=points,
325
+ )
326
+
327
+
328
+ def score_bp(matched_terms: int, total_terms: int, matched_values: int, total_values: int) -> int:
329
+ """Return ``floor(score * 10_000)`` using integer arithmetic only.
330
+
331
+ ``score = 4/5 * a/b + 1/5 * c/d`` is ``(4ad + cb) / (5bd)``, so one integer floor division
332
+ gives the identical result the rational form would, without constructing a
333
+ :class:`~fractions.Fraction` per scored line. ``exact_score`` keeps the rational definition
334
+ available, and a test pins the two together across a grid of inputs.
335
+ """
336
+
337
+ terms = max(1, total_terms)
338
+ if total_values:
339
+ numerator = 10_000 * (4 * matched_terms * total_values + matched_values * terms)
340
+ return numerator // (5 * terms * total_values)
341
+ return (10_000 * matched_terms) // terms
342
+
343
+
344
+ def exact_score(
345
+ matched_terms: int, total_terms: int, matched_values: int, total_values: int
346
+ ) -> Fraction:
347
+ """The rational definition of the score, kept as the reference for :func:`score_bp`."""
348
+
349
+ term_coverage = Fraction(matched_terms, max(1, total_terms))
350
+ if not total_values:
351
+ return term_coverage
352
+ value_coverage = Fraction(matched_values, total_values)
353
+ return Fraction(4, 5) * term_coverage + Fraction(1, 5) * value_coverage
354
+
355
+
356
+ def values(text: str) -> frozenset[str]:
357
+ """Extract the numbers, comparison operators, and identifiers a claim commits to."""
358
+
359
+ found: set[str] = set(_NUMBER.findall(text))
360
+ remaining = text
361
+ for symbol in _SYMBOL_OPERATORS:
362
+ if symbol in remaining:
363
+ found.add(_CANONICAL_OPERATOR.get(symbol, symbol))
364
+ remaining = remaining.replace(symbol, " ")
365
+ found.update(match.group(0).strip("`").casefold() for match in _IDENTIFIER.finditer(text))
366
+ return frozenset(found)
367
+
368
+
369
+ def segment(draft_text: str, suffix: str) -> list[Claim]:
370
+ """Split a draft into located claims, marking connective prose ``skipped`` later.
371
+
372
+ Markdown and text become sentences with their starting line numbers; fenced blocks become one
373
+ code claim each; display math becomes a formula claim; numbered list items become steps. A
374
+ notebook contributes one claim per code cell and its markdown cells segmented as prose, with
375
+ line numbers running over the concatenated cell sources so a citation still locates the text.
376
+ """
377
+
378
+ if suffix.casefold() == ".ipynb":
379
+ return _segment_notebook(draft_text)
380
+ return _segment_text(draft_text)
381
+
382
+
383
+ def retrieve(
384
+ claim: Claim,
385
+ sources: Sequence[ScanSource] | LineIndex,
386
+ top: int = TOP_CITATIONS,
387
+ ) -> list[Citation]:
388
+ """Return the top course spans for one claim, highest score first."""
389
+
390
+ index = sources if isinstance(sources, LineIndex) else LineIndex(sources)
391
+ return index.retrieve(claim, top=top)
392
+
393
+
394
+ def classify(claim: Claim, candidates: Sequence[Citation]) -> Finding:
395
+ """Assign exactly one evidence status, deterministically and lexically."""
396
+
397
+ if not _is_checkable(claim):
398
+ return Finding(claim, "skipped", [], "connective prose, not a checkable claim")
399
+
400
+ above = [item for item in candidates if item.retrieval_score_bp >= CANDIDATE_FLOOR_BP]
401
+ if not above:
402
+ nearest = candidates[0] if candidates else None
403
+ note = "no matching evidence was found in the scanned course material"
404
+ if nearest is not None:
405
+ note = (
406
+ f"{note}; nearest below-threshold span (not evidence): "
407
+ f"{nearest.path}:{nearest.line}"
408
+ )
409
+ return Finding(claim, "no_matching_evidence", [], note)
410
+
411
+ best = above[0]
412
+ if best.retrieval_score_bp >= STRONG_MATCH_FLOOR_BP:
413
+ if _possible_conflict(claim.text, best.excerpt):
414
+ return Finding(
415
+ claim,
416
+ "possible_conflict",
417
+ above,
418
+ "your materials may say something different — compare both spans yourself",
419
+ )
420
+ claim_values = values(claim.text)
421
+ if not claim_values or claim_values <= values(best.excerpt):
422
+ return Finding(claim, "evidence_found", above, None)
423
+ missing = ", ".join(sorted(claim_values - values(best.excerpt)))
424
+ return Finding(
425
+ claim,
426
+ "related_evidence",
427
+ above,
428
+ f"related, not asserted; not found in the cited span: {missing}",
429
+ )
430
+ return Finding(
431
+ claim,
432
+ "related_evidence",
433
+ above,
434
+ "related, not asserted; it does not establish this specific claim",
435
+ )
436
+
437
+
438
+ def check(draft: Path, course_dir: Path, *, assignment: str | None = None) -> CheckReport:
439
+ """Scan one draft against one course's verified material.
440
+
441
+ An empty source set is an error rather than a clean bill of health: a scan that found nothing
442
+ to read has not checked anything.
443
+ """
444
+
445
+ draft_path = Path(draft).expanduser().absolute()
446
+ text = _read_text(draft_path)
447
+ if text is None:
448
+ raise A2LError("draft file is unreadable")
449
+ suffix = draft_path.suffix.casefold()
450
+ if suffix and suffix not in SUPPORTED_SUFFIXES:
451
+ raise A2LError(f"check does not read {suffix} drafts")
452
+
453
+ vault = _vault_for(course_dir)
454
+ scope, sources = _scan_sources(vault, course_dir, draft_path, assignment)
455
+ if not sources:
456
+ # "Nothing verified in this course" is a sync problem. "Nothing verified matched this
457
+ # assignment" is not, and sending the student to `a2l sync` for it would be a false next
458
+ # action that hides the real remedy: scan the whole course.
459
+ if scope != "whole course" and ground.verified_sources(vault, course_dir):
460
+ raise A2LError(
461
+ f"no verified course material matched the assignment {scope!r}, so there is "
462
+ "nothing to scan it against. Scan the whole course instead: move the draft out "
463
+ "of the assignment folder, or omit --assignment."
464
+ )
465
+ raise A2LError("this course has no verified material to scan yet; run: a2l sync")
466
+
467
+ index = LineIndex(sources)
468
+ if not len(index):
469
+ raise A2LError("the verified course material contains no readable lines; re-run: a2l sync")
470
+
471
+ findings = tuple(classify(claim, index.retrieve(claim)) for claim in segment(text, suffix))
472
+ code, _name = _course_identity(course_dir)
473
+ return CheckReport(
474
+ draft=_display(draft_path, vault.root),
475
+ course=paths.rel_posix(course_dir, vault.root),
476
+ course_code=code,
477
+ scope=scope,
478
+ findings=findings,
479
+ coverage_gaps=_coverage_gaps(course_dir, findings),
480
+ notation=_notation(findings, index),
481
+ revisions={
482
+ source.citation_path: {
483
+ "source_sha256": source.source_sha256,
484
+ "derived_sha256": source.derived_sha256,
485
+ }
486
+ for source in sources
487
+ },
488
+ )
489
+
490
+
491
+ def render(report: CheckReport) -> str:
492
+ """Render the report for a human, leading with the experimental disclosure."""
493
+
494
+ counts = _counts(report)
495
+ lines = [
496
+ DISCLOSURE,
497
+ "",
498
+ f"{report.course_code} · {report.scope} · {report.draft}",
499
+ (
500
+ f"{len(report.findings)} claims · {counts['evidence_found']} with matching evidence · "
501
+ f"{counts['related_evidence']} related · {counts['no_matching_evidence']} no match · "
502
+ f"{counts['possible_conflict']} to compare · {counts['skipped']} skipped"
503
+ ),
504
+ "",
505
+ ]
506
+ marks = {
507
+ "evidence_found": "+",
508
+ "related_evidence": "~",
509
+ "no_matching_evidence": "x",
510
+ "possible_conflict": "?",
511
+ "skipped": "-",
512
+ }
513
+ for finding in report.findings:
514
+ if finding.status == "skipped":
515
+ continue
516
+ lines.append(f"{marks[finding.status]} L{finding.claim.line} {finding.claim.text}")
517
+ if finding.note:
518
+ lines.append(f" {finding.note}")
519
+ for citation in finding.citations:
520
+ lines.append(f" {citation.path}:{citation.line}")
521
+ lines.append(f" source excerpt: {citation.excerpt}")
522
+ lines.append("")
523
+
524
+ if report.notation:
525
+ lines.extend(["NOTATION", ""])
526
+ for item in report.notation:
527
+ if item.candidate and item.citation is not None:
528
+ lines.append(
529
+ f'· you write "{item.term}"; nearest course wording candidate: '
530
+ f'"{item.candidate}" ({item.citation.path}:{item.citation.line})'
531
+ )
532
+ else:
533
+ lines.append(f'· you write "{item.term}"; no close course wording was found')
534
+ lines.append("")
535
+
536
+ if report.coverage_gaps:
537
+ lines.extend(["COVERAGE", ""])
538
+ for gap in report.coverage_gaps:
539
+ action = f" → {gap.fetch_command}" if gap.fetch_command else ""
540
+ lines.append(f"· {gap.title}: {gap.note}{action}")
541
+ lines.append("")
542
+
543
+ lines.extend(
544
+ [
545
+ "─" * 53,
546
+ (
547
+ "A status here describes what a lexical scan matched. It is not proof that your "
548
+ "work is right or wrong, and it says nothing about grading or academic policy. "
549
+ "Read the cited sources yourself."
550
+ ),
551
+ "",
552
+ ]
553
+ )
554
+ return "\n".join(lines)
555
+
556
+
557
+ def render_json(report: CheckReport) -> str:
558
+ """Render the same content as stable machine-readable JSON."""
559
+
560
+ payload = {
561
+ "check_algorithm_version": report.algorithm_version,
562
+ "disclosure": DISCLOSURE,
563
+ "not_proof": (
564
+ "A status describes a lexical match only. It is not proof of correctness, "
565
+ "incorrectness, policy compliance, or academic integrity."
566
+ ),
567
+ "draft": report.draft,
568
+ "course": report.course,
569
+ "course_code": report.course_code,
570
+ "scope": report.scope,
571
+ "candidate_floor_bp": CANDIDATE_FLOOR_BP,
572
+ "strong_match_floor_bp": STRONG_MATCH_FLOOR_BP,
573
+ "findings": [
574
+ {
575
+ "line": finding.claim.line,
576
+ "text": finding.claim.text,
577
+ "kind": finding.claim.kind,
578
+ "status": finding.status,
579
+ "score_bp": (finding.citations[0].retrieval_score_bp if finding.citations else 0),
580
+ "citations": [
581
+ {
582
+ "path": citation.path,
583
+ "line": citation.line,
584
+ "excerpt": citation.excerpt,
585
+ "source_sha256": citation.source_sha256,
586
+ "derived_sha256": citation.derived_sha256,
587
+ "retrieval_score_bp": citation.retrieval_score_bp,
588
+ }
589
+ for citation in finding.citations
590
+ ],
591
+ "note": finding.note,
592
+ }
593
+ for finding in report.findings
594
+ ],
595
+ "coverage_gaps": [
596
+ {
597
+ "source_key": gap.source_key,
598
+ "source_id": gap.source_id,
599
+ "title": gap.title,
600
+ "availability": gap.availability,
601
+ "note": gap.note,
602
+ "fetch_command": gap.fetch_command,
603
+ }
604
+ for gap in report.coverage_gaps
605
+ ],
606
+ "notation": [
607
+ {
608
+ "term": item.term,
609
+ "candidate": item.candidate,
610
+ "citation": None
611
+ if item.citation is None
612
+ else {"path": item.citation.path, "line": item.citation.line},
613
+ }
614
+ for item in report.notation
615
+ ],
616
+ "revisions": report.revisions,
617
+ }
618
+ return json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n"
619
+
620
+
621
+ def _segment_text(text: str) -> list[Claim]:
622
+ claims: list[Claim] = []
623
+ block: list[tuple[int, str]] = []
624
+ fence: list[str] = []
625
+ fence_line = 0
626
+ fence_kind: ClaimKind = "code"
627
+ inside = False
628
+
629
+ for number, raw in enumerate(text.splitlines(), start=1):
630
+ stripped = raw.strip()
631
+ opener = _fence_kind(stripped)
632
+ if inside:
633
+ if _closes_fence(stripped, fence_kind):
634
+ claims.append(Claim(fence_line, "\n".join(fence).strip(), fence_kind))
635
+ inside = False
636
+ fence = []
637
+ continue
638
+ fence.append(raw)
639
+ continue
640
+ if opener is not None:
641
+ _emit_block(block, claims)
642
+ block = []
643
+ inside = True
644
+ fence_line = number
645
+ fence_kind = opener
646
+ fence = []
647
+ continue
648
+ if not stripped or stripped.startswith("<!--"):
649
+ _emit_block(block, claims)
650
+ block = []
651
+ continue
652
+ text_only = _MARKDOWN_IMAGE.sub(" ", stripped.lstrip("#")).strip()
653
+ text_only = re.sub(r"\s{2,}", " ", text_only)
654
+ if not text_only:
655
+ # An image-only line is structure, like a heading marker or a page comment.
656
+ _emit_block(block, claims)
657
+ block = []
658
+ continue
659
+ block.append((number, text_only))
660
+
661
+ if inside and fence:
662
+ claims.append(Claim(fence_line, "\n".join(fence).strip(), fence_kind))
663
+ _emit_block(block, claims)
664
+ return claims
665
+
666
+
667
+ def _emit_block(block: Sequence[tuple[int, str]], claims: list[Claim]) -> None:
668
+ """Emit one block, keeping an enumerated derivation step intact.
669
+
670
+ A numbered item is one checkable unit, so it is never sentence-split: splitting after the
671
+ enumerator would leave ``1.`` as its own meaningless claim.
672
+ """
673
+
674
+ if not block:
675
+ return
676
+ segments: list[list[tuple[int, str]]] = []
677
+ current: list[tuple[int, str]] = []
678
+ for number, text in block:
679
+ if _STEP.match(text) and current:
680
+ segments.append(current)
681
+ current = []
682
+ current.append((number, text))
683
+ if current:
684
+ segments.append(current)
685
+
686
+ for segment_lines in segments:
687
+ start, first = segment_lines[0]
688
+ if _STEP.match(first):
689
+ joined = " ".join(value for _number, value in segment_lines)
690
+ claims.append(Claim(start, joined, "step"))
691
+ continue
692
+ _flush(segment_lines, claims)
693
+
694
+
695
+ def _segment_notebook(text: str) -> list[Claim]:
696
+ try:
697
+ raw = json.loads(text)
698
+ except json.JSONDecodeError as exc:
699
+ raise A2LError("notebook draft is not valid JSON") from exc
700
+ cells = raw.get("cells") if isinstance(raw, Mapping) else None
701
+ if not isinstance(cells, list):
702
+ raise A2LError("notebook draft has no cells")
703
+ claims: list[Claim] = []
704
+ line = 1
705
+ for cell in cells:
706
+ if not isinstance(cell, Mapping):
707
+ continue
708
+ body = _cell_source(cell.get("source"))
709
+ height = max(1, len(body.splitlines()))
710
+ if cell.get("cell_type") == "code":
711
+ if body.strip():
712
+ claims.append(Claim(line, body.strip(), "code"))
713
+ else:
714
+ for claim in _segment_text(body):
715
+ claims.append(Claim(line + claim.line - 1, claim.text, claim.kind))
716
+ line += height
717
+ return claims
718
+
719
+
720
+ def _cell_source(value: object) -> str:
721
+ if isinstance(value, str):
722
+ return value
723
+ if isinstance(value, list):
724
+ return "".join(str(part) for part in value)
725
+ return ""
726
+
727
+
728
+ def _fence_kind(stripped: str) -> ClaimKind | None:
729
+ if stripped.startswith("```") or stripped.startswith("~~~"):
730
+ return "code"
731
+ if stripped == "$$" or stripped.startswith("\\["):
732
+ return "formula"
733
+ return None
734
+
735
+
736
+ def _closes_fence(stripped: str, kind: ClaimKind) -> bool:
737
+ if kind == "code":
738
+ return stripped.startswith("```") or stripped.startswith("~~~")
739
+ return stripped == "$$" or stripped.startswith("\\]")
740
+
741
+
742
+ def _flush(paragraph: Sequence[tuple[int, str]], claims: list[Claim]) -> None:
743
+ if not paragraph:
744
+ return
745
+ joined = ""
746
+ offsets: list[tuple[int, int]] = []
747
+ for number, piece in paragraph:
748
+ if joined:
749
+ joined += " "
750
+ offsets.append((len(joined), number))
751
+ joined += piece
752
+ cursor = 0
753
+ for sentence in _SENTENCE_SPLIT.split(joined):
754
+ text = sentence.strip()
755
+ if not text:
756
+ continue
757
+ start = joined.find(text, cursor)
758
+ if start < 0:
759
+ start = cursor
760
+ cursor = start + len(text)
761
+ claims.append(Claim(_line_for(start, offsets), text, _kind(text)))
762
+
763
+
764
+ def _line_for(offset: int, offsets: Sequence[tuple[int, int]]) -> int:
765
+ number = offsets[0][1] if offsets else 1
766
+ for start, candidate in offsets:
767
+ if start <= offset:
768
+ number = candidate
769
+ else:
770
+ break
771
+ return number
772
+
773
+
774
+ def _kind(text: str) -> ClaimKind:
775
+ if _STEP.match(text):
776
+ return "step"
777
+ words = [token for token in _RUN.findall(text.casefold()) if not token.isdigit()]
778
+ if _NUMBER_OR_MATH.search(text) and len(words) < 6:
779
+ return "formula"
780
+ return "prose"
781
+
782
+
783
+ def _is_checkable(claim: Claim) -> bool:
784
+ """Return whether a claim commits to something a lexical scan can look for.
785
+
786
+ Explicit and versioned on purpose: this does not parse noun phrases or judge meaning. A claim
787
+ counts as checkable when it carries a number or math symbol, a definition cue, a code or API
788
+ identifier, a named-method cue, or at least three content tokens once function and coursework
789
+ words are removed.
790
+ """
791
+
792
+ if claim.kind in {"code", "formula", "step"}:
793
+ return True
794
+ text = claim.text
795
+ if _NUMBER_OR_MATH.search(text):
796
+ return True
797
+ if _DEFINITION_CUE.search(text):
798
+ return True
799
+ if _IDENTIFIER.search(text):
800
+ return True
801
+ if _NAMED_METHOD.search(text):
802
+ return True
803
+ return len(_content_tokens(text)) >= 3
804
+
805
+
806
+ def _content_tokens(text: str) -> list[str]:
807
+ return [token for token in tok(text) if token not in GENERIC and token not in _FUNCTION]
808
+
809
+
810
+ def _possible_conflict(claim_text: str, source_text: str) -> bool:
811
+ """Return whether two spans match one of exactly two allowlisted conflict templates.
812
+
813
+ Both templates require the token sequences to be otherwise identical, which is what keeps the
814
+ status narrow. A differing number changes the token sequence, so it can never qualify.
815
+ """
816
+
817
+ claim_ops, claim_core = _relation_signature(claim_text)
818
+ source_ops, source_core = _relation_signature(source_text)
819
+ if (
820
+ claim_core
821
+ and claim_core == source_core
822
+ and len(claim_ops) == len(source_ops) >= 1
823
+ and all(
824
+ other in _OPPOSITES.get(mine, frozenset())
825
+ for mine, other in zip(claim_ops, source_ops, strict=True)
826
+ )
827
+ ):
828
+ return True
829
+
830
+ claim_negated, claim_tokens = _polarity(claim_text)
831
+ source_negated, source_tokens = _polarity(source_text)
832
+ return bool(claim_tokens) and claim_tokens == source_tokens and claim_negated != source_negated
833
+
834
+
835
+ def _relation_signature(text: str) -> tuple[tuple[str, ...], tuple[str, ...]]:
836
+ """Split a span into its ordered comparison operators and the tokens around them.
837
+
838
+ Scanning left to right, longest operator first, keeps ``<=`` from being read as ``<`` and
839
+ keeps the operator order faithful to the sentence.
840
+ """
841
+
842
+ normalized = text
843
+ for pattern, replacement in _WORD_OPERATORS:
844
+ normalized = pattern.sub(f" {replacement} ", normalized)
845
+ operators: list[str] = []
846
+ remainder: list[str] = []
847
+ position = 0
848
+ while position < len(normalized):
849
+ for symbol in _SYMBOL_OPERATORS:
850
+ if normalized.startswith(symbol, position):
851
+ operators.append(_CANONICAL_OPERATOR.get(symbol, symbol))
852
+ remainder.append(" ")
853
+ position += len(symbol)
854
+ break
855
+ else:
856
+ remainder.append(normalized[position])
857
+ position += 1
858
+ return tuple(operators), tuple(tok("".join(remainder)))
859
+
860
+
861
+ def _polarity(text: str) -> tuple[bool, tuple[str, ...]]:
862
+ """Return the narrow ``is``/``is not`` predicate polarity template.
863
+
864
+ Broad lexical negation (``never``, ``cannot``, ``without``, and friends) is deliberately not a
865
+ conflict signal. The design allowlist only treats a single explicit ``is`` predicate and its
866
+ immediately following ``not`` as opposite polarity; any other negation or multiple predicates
867
+ makes the template inapplicable.
868
+ """
869
+
870
+ matches = tuple(_IS_POLARITY.finditer(text))
871
+ if len(matches) != 1:
872
+ return False, ()
873
+ match = matches[0]
874
+ stripped = text[: match.start()] + " " + text[match.end() :]
875
+ if _NEGATION.search(stripped) is not None:
876
+ return False, ()
877
+ return bool(match.group("not")), tuple(tok(stripped))
878
+
879
+
880
+ def _vault_for(course_dir: Path) -> Vault:
881
+ candidate = Path(course_dir).expanduser()
882
+ for parent in (candidate, *candidate.parents):
883
+ try:
884
+ if Vault.is_vault(parent):
885
+ return Vault(parent)
886
+ except (OSError, ValueError):
887
+ continue
888
+ raise A2LError("this course is not inside an Agent2Learn vault; run: a2l init")
889
+
890
+
891
+ def _scan_sources(
892
+ vault: Vault,
893
+ course_dir: Path,
894
+ draft: Path,
895
+ assignment: str | None,
896
+ ) -> tuple[str, tuple[ScanSource, ...]]:
897
+ item = assignment or _assignment_for(course_dir, draft)
898
+ if item is not None:
899
+ selected = ground.select_sources(vault, course_dir, item)
900
+ scope = ground.assignment_title(course_dir, item, fallback=item)
901
+ else:
902
+ selected = ground.verified_sources(vault, course_dir)
903
+ scope = "whole course"
904
+ excluded = _same_file_key(draft)
905
+ sources = tuple(
906
+ ScanSource(
907
+ path=vault.root / PurePosixPath(source.citation_path),
908
+ citation_path=source.citation_path,
909
+ source_sha256=source.source_sha256,
910
+ derived_sha256=source.derived_sha256,
911
+ )
912
+ for source in selected
913
+ if _same_file_key(vault.root / PurePosixPath(source.citation_path)) != excluded
914
+ )
915
+ return scope, sources
916
+
917
+
918
+ def _assignment_for(course_dir: Path, draft: Path) -> str | None:
919
+ """Return the assignment folder name when the draft sits inside one."""
920
+
921
+ assignments = _same_file_key(course_dir / "assignments")
922
+ previous: Path | None = None
923
+ for parent in Path(draft).expanduser().parents:
924
+ if _same_file_key(parent) == assignments and previous is not None:
925
+ return previous.name
926
+ previous = parent
927
+ return None
928
+
929
+
930
+ def _coverage_gaps(course_dir: Path, findings: Sequence[Finding]) -> tuple[CoverageGap, ...]:
931
+ """Report unscannable material whose title shares a term with an unresolved claim.
932
+
933
+ Reported before a reader treats ``no_matching_evidence`` as absence: the material may simply
934
+ not be on disk yet.
935
+ """
936
+
937
+ wanted: set[str] = set()
938
+ for finding in findings:
939
+ if finding.status in {"no_matching_evidence", "related_evidence"}:
940
+ wanted.update(ground.distinguishing_terms(finding.claim.text))
941
+ if not wanted:
942
+ return ()
943
+ rows = course_index.read_content_map(course_dir)["topics"]
944
+ if not isinstance(rows, list):
945
+ return ()
946
+ gaps: dict[str, CoverageGap] = {}
947
+ for row in rows:
948
+ if not isinstance(row, Mapping) or row.get("path"):
949
+ continue
950
+ source_key = row.get("source_key")
951
+ source_id = row.get("source_id")
952
+ if not isinstance(source_key, str) or not isinstance(source_id, str):
953
+ continue
954
+ title = str(row.get("title") or source_key)
955
+ if not ground.distinguishing_terms(title) & wanted:
956
+ continue
957
+ availability = str(row.get("availability") or "metadata_only")
958
+ gaps[source_key] = CoverageGap(
959
+ source_key=source_key,
960
+ source_id=source_id,
961
+ title=title,
962
+ availability=availability,
963
+ note=_AVAILABILITY_NOTES.get(availability, "unavailable locally"),
964
+ fetch_command=(f"a2l fetch {source_id}" if availability in _FETCHABLE else None),
965
+ )
966
+ return tuple(gaps[key] for key in sorted(gaps))
967
+
968
+
969
+ def _notation(findings: Sequence[Finding], index: LineIndex) -> tuple[NotationCandidate, ...]:
970
+ """Flag draft terms absent from the scanned material, with a candidate, never a correction."""
971
+
972
+ missing: dict[str, None] = {}
973
+ for finding in findings:
974
+ if finding.status == "skipped":
975
+ continue
976
+ for token in _content_tokens(finding.claim.text):
977
+ if len(token) < 4 or token.isdigit() or token in index.vocabulary:
978
+ continue
979
+ missing.setdefault(token, None)
980
+ candidates: list[NotationCandidate] = []
981
+ for term in sorted(missing):
982
+ best_term: str | None = None
983
+ best_ratio = 0
984
+ for known in sorted(index.vocabulary):
985
+ ratio = int(SequenceMatcher(None, term, known).ratio() * 10_000)
986
+ if ratio > best_ratio:
987
+ best_ratio, best_term = ratio, known
988
+ if best_term is None or best_ratio < NOTATION_FLOOR_BP:
989
+ candidates.append(NotationCandidate(term, None, None))
990
+ continue
991
+ path, line = index.vocabulary[best_term]
992
+ candidates.append(
993
+ NotationCandidate(
994
+ term,
995
+ best_term,
996
+ Citation(
997
+ path=path,
998
+ line=line,
999
+ excerpt=best_term,
1000
+ source_sha256="",
1001
+ derived_sha256="",
1002
+ retrieval_score_bp=best_ratio,
1003
+ ),
1004
+ )
1005
+ )
1006
+ return tuple(candidates)
1007
+
1008
+
1009
+ def _counts(report: CheckReport) -> dict[str, int]:
1010
+ counts = dict.fromkeys(
1011
+ (
1012
+ "evidence_found",
1013
+ "related_evidence",
1014
+ "no_matching_evidence",
1015
+ "possible_conflict",
1016
+ "skipped",
1017
+ ),
1018
+ 0,
1019
+ )
1020
+ for finding in report.findings:
1021
+ counts[finding.status] += 1
1022
+ return counts
1023
+
1024
+
1025
+ def _course_identity(course_dir: Path) -> tuple[str, str]:
1026
+ rows = course_index.read_content_map(course_dir)["topics"]
1027
+ if isinstance(rows, list):
1028
+ for row in rows:
1029
+ if not isinstance(row, Mapping):
1030
+ continue
1031
+ code = row.get("course_code")
1032
+ name = row.get("course_name")
1033
+ if isinstance(code, str) and code:
1034
+ return code, name if isinstance(name, str) and name else code
1035
+ return course_dir.name, course_dir.name
1036
+
1037
+
1038
+ def _display(path: Path, root: Path) -> str:
1039
+ try:
1040
+ return paths.rel_posix(path, root)
1041
+ except (ValueError, OSError):
1042
+ return path.name
1043
+
1044
+
1045
+ def _same_file_key(path: Path) -> tuple[int, int] | str:
1046
+ candidate = Path(path).expanduser()
1047
+ try:
1048
+ identity = paths.long_path(candidate).stat()
1049
+ if identity.st_ino:
1050
+ return identity.st_dev, identity.st_ino
1051
+ except OSError:
1052
+ pass
1053
+ return os.path.normcase(os.path.abspath(paths.plain_path(candidate)))
1054
+
1055
+
1056
+ def _read_text(path: Path) -> str | None:
1057
+ try:
1058
+ with open(
1059
+ os.fspath(paths.long_path(path)), encoding="utf-8", errors="ignore", newline=""
1060
+ ) as handle:
1061
+ return handle.read()
1062
+ except (FileNotFoundError, IsADirectoryError, OSError, UnicodeError):
1063
+ return None
1064
+
1065
+
1066
+ __all__ = [
1067
+ "CANDIDATE_FLOOR_BP",
1068
+ "CHECK_ALGORITHM_VERSION",
1069
+ "DISCLOSURE",
1070
+ "NOTATION_FLOOR_BP",
1071
+ "STRONG_MATCH_FLOOR_BP",
1072
+ "SUPPORTED_SUFFIXES",
1073
+ "TOP_CITATIONS",
1074
+ "CheckReport",
1075
+ "Citation",
1076
+ "Claim",
1077
+ "CoverageGap",
1078
+ "Finding",
1079
+ "LineIndex",
1080
+ "NotationCandidate",
1081
+ "ScanSource",
1082
+ "check",
1083
+ "classify",
1084
+ "render",
1085
+ "render_json",
1086
+ "retrieve",
1087
+ "exact_score",
1088
+ "score_bp",
1089
+ "segment",
1090
+ "values",
1091
+ ]