lorecraft 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. lorecraft/__init__.py +5 -0
  2. lorecraft/__main__.py +6 -0
  3. lorecraft/checks/__init__.py +35 -0
  4. lorecraft/checks/budget.py +54 -0
  5. lorecraft/checks/database.py +124 -0
  6. lorecraft/checks/header.py +122 -0
  7. lorecraft/checks/reporting.py +59 -0
  8. lorecraft/checks/run.py +230 -0
  9. lorecraft/checks/structure.py +194 -0
  10. lorecraft/cli/__init__.py +9 -0
  11. lorecraft/cli/app.py +68 -0
  12. lorecraft/cli/check_run.py +194 -0
  13. lorecraft/cli/commands/__init__.py +8 -0
  14. lorecraft/cli/commands/check/__init__.py +86 -0
  15. lorecraft/cli/commands/check/budget.py +70 -0
  16. lorecraft/cli/commands/check/header.py +71 -0
  17. lorecraft/cli/commands/check/structure.py +70 -0
  18. lorecraft/cli/commands/inspect.py +52 -0
  19. lorecraft/cli/commands/version.py +22 -0
  20. lorecraft/cli/registry.py +111 -0
  21. lorecraft/cli/root.py +71 -0
  22. lorecraft/cli/select.py +102 -0
  23. lorecraft/cli/version.py +106 -0
  24. lorecraft/cli/workspace_tree.py +144 -0
  25. lorecraft/core/__init__.py +8 -0
  26. lorecraft/core/error.py +5 -0
  27. lorecraft/metadata.py +25 -0
  28. lorecraft/project/__init__.py +6 -0
  29. lorecraft/project/aspect/__init__.py +28 -0
  30. lorecraft/project/aspect/filename.py +53 -0
  31. lorecraft/project/aspect/name.py +96 -0
  32. lorecraft/project/aspect/namespace.py +90 -0
  33. lorecraft/project/corpus/__init__.py +15 -0
  34. lorecraft/project/corpus/name.py +96 -0
  35. lorecraft/project/document/__init__.py +23 -0
  36. lorecraft/project/document/ref.py +31 -0
  37. lorecraft/project/document/repo.py +147 -0
  38. lorecraft/project/layout.py +17 -0
  39. lorecraft/project/schemas/__init__.py +66 -0
  40. lorecraft/project/schemas/header.py +67 -0
  41. lorecraft/project/schemas/name.py +34 -0
  42. lorecraft/project/schemas/repo.py +146 -0
  43. lorecraft/project/schemas/spec_file.py +184 -0
  44. lorecraft/project/schemas/structure.py +280 -0
  45. lorecraft/project/schemas/structure_file.py +130 -0
  46. lorecraft/project/syntax/__init__.py +33 -0
  47. lorecraft/project/syntax/document.py +200 -0
  48. lorecraft/project/syntax/frontmatter.py +120 -0
  49. lorecraft/project/syntax/heading.py +34 -0
  50. lorecraft/project/syntax/o200k_base.tiktoken +199998 -0
  51. lorecraft/project/syntax/position.py +43 -0
  52. lorecraft/project/syntax/tokens.py +76 -0
  53. lorecraft/project/workspace/__init__.py +13 -0
  54. lorecraft/project/workspace/loader.py +192 -0
  55. lorecraft/project/workspace/model.py +214 -0
  56. lorecraft/vfs/__init__.py +47 -0
  57. lorecraft/vfs/changes.py +84 -0
  58. lorecraft/vfs/disk.py +323 -0
  59. lorecraft/vfs/path.py +107 -0
  60. lorecraft/vfs/snapshot.py +236 -0
  61. lorecraft/vfs/view.py +160 -0
  62. lorecraft-0.1.0.dist-info/METADATA +165 -0
  63. lorecraft-0.1.0.dist-info/RECORD +67 -0
  64. lorecraft-0.1.0.dist-info/WHEEL +4 -0
  65. lorecraft-0.1.0.dist-info/entry_points.txt +3 -0
  66. lorecraft-0.1.0.dist-info/licenses/LICENSE-APACHE +200 -0
  67. lorecraft-0.1.0.dist-info/licenses/LICENSE-MIT +21 -0
lorecraft/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """Lorecraft: checks for a repository's agent-facing documentation."""
2
+
3
+ from .metadata import __version__
4
+
5
+ __all__: list[str] = ['__version__']
lorecraft/__main__.py ADDED
@@ -0,0 +1,6 @@
1
+ """Entry point for `python -m lorecraft`, equivalent to the `lorecraft` console script."""
2
+
3
+ from .cli import main
4
+
5
+ if __name__ == '__main__':
6
+ main()
@@ -0,0 +1,35 @@
1
+ """Document checks: each validates one aspect of the documents a workspace model lists.
2
+
3
+ A check is pure over the one part of a document it reads: the header check over the frontmatter node, the
4
+ structure check over the headings, the budget check over the token count. The ``Database`` caches the model,
5
+ the frontmatter, the parse trees and the token counts of one snapshot, and every check reads through it, so each
6
+ part is computed once whichever checks read it; ``run`` asks the database for the part the check reads and hands
7
+ only that to the check. A check returns violations, which name no document; ``run`` files them under the
8
+ document's report, which locates them as findings. Nothing here prints: the ``check`` commands own the output
9
+ and the exit codes.
10
+ """
11
+
12
+ from .budget import BudgetCheckResult, validate_budget
13
+ from .database import Database
14
+ from .header import HeaderCheckResult, validate_header
15
+ from .reporting import Finding, Violation, format_finding
16
+ from .run import CheckRun, DocumentReport, run_budget, run_header, run_structure
17
+ from .structure import StructureCheckResult, validate_structure
18
+
19
+ __all__ = [
20
+ 'Database',
21
+ 'Finding',
22
+ 'Violation',
23
+ 'format_finding',
24
+ 'HeaderCheckResult',
25
+ 'validate_header',
26
+ 'StructureCheckResult',
27
+ 'validate_structure',
28
+ 'BudgetCheckResult',
29
+ 'validate_budget',
30
+ 'DocumentReport',
31
+ 'CheckRun',
32
+ 'run_header',
33
+ 'run_structure',
34
+ 'run_budget',
35
+ ]
@@ -0,0 +1,54 @@
1
+ """Check one document's whole-file token count against the token budgets that govern it.
2
+
3
+ The check is pure: it takes the already validated structure aspects that govern a document and the document's
4
+ token count, and returns violations. It reads the count and nothing else, so it is handed that number rather than
5
+ the text, and not the document's path: the run that called it attaches that. The budget is the global ``tokens``
6
+ key of a structure specification, but it is a check of its own: the structure check reads a document's parse tree,
7
+ and this one reads its raw text, frontmatter, code and tables included, since that is what loading it costs an
8
+ agent.
9
+
10
+ Each aspect is applied on its own. A namespace specification's budget does not replace the corpus one, so a
11
+ document governed by both must fit both, and a namespace can only tighten the corpus budget.
12
+ """
13
+
14
+ from dataclasses import dataclass
15
+ from typing import Final
16
+
17
+ from lorecraft.project.schemas import StructureAspect
18
+ from lorecraft.project.syntax import LineNumber
19
+
20
+ from .reporting import Violation
21
+
22
+ _FIRST_LINE: Final[LineNumber] = LineNumber(1)
23
+ """Where a budget violation is reported: it concerns the whole file, not one line of it."""
24
+
25
+
26
+ @dataclass(frozen=True, slots=True)
27
+ class BudgetCheckResult:
28
+ """What the budget check found in one document.
29
+
30
+ Attributes:
31
+ violations: One per aspect whose budget the document exceeds, in aspect order; empty when it fits all.
32
+ """
33
+
34
+ violations: tuple[Violation, ...]
35
+
36
+
37
+ def validate_budget(aspects: tuple[StructureAspect, ...], *, token_count: int) -> BudgetCheckResult:
38
+ """Check one document's token count against the budget of each structure aspect that sets one.
39
+
40
+ Every violation's message ends by naming the structure specification file that sets the budget, as the header
41
+ check names its schema: the number is in that file, not in the prose.
42
+
43
+ Args:
44
+ aspects: Applied each on its own; an aspect without a ``tokens`` budget is skipped. Empty means the
45
+ document is ungoverned, which yields no violations.
46
+ token_count: The ``o200k_base`` tokens in the document's whole file, as ``count_tokens`` counts them.
47
+ """
48
+ violations: list[Violation] = []
49
+ for aspect in aspects:
50
+ if aspect.tokens is None or token_count <= aspect.tokens:
51
+ continue
52
+ message = f'{token_count} tokens; the budget is {aspect.tokens} (per {aspect.path.name})'
53
+ violations.append(Violation(line=_FIRST_LINE, rule='budget.tokens', message=message, spec=aspect.path))
54
+ return BudgetCheckResult(violations=tuple(violations))
@@ -0,0 +1,124 @@
1
+ """One frozen state of a workspace: its snapshot, and the cached data every check reads through.
2
+
3
+ The layering follows the IntelliJ Platform's. The ``Snapshot`` plays the virtual file system: every byte a
4
+ check can see, and never changed once built. Above it sit two kinds of cached data, each computed on first
5
+ use and kept for as long as the database lives (pattern-memoization):
6
+
7
+ - ``model()``: the workspace model, like the IDE's project model. It reads the structure of the snapshot, the
8
+ specifications and the corpus directories, and no document's contents.
9
+ - ``frontmatter(ref)``: one document's frontmatter node, like a stub: the part of a file the IDE reads without
10
+ building its full syntax tree. It reads that document's bytes and nothing else.
11
+ - ``parse(ref)``: one document's parse tree, like a PSI file or a per-file index entry. It reads that
12
+ document's bytes and nothing else.
13
+ - ``tokens(ref)``: what one document's whole file costs an agent that loads it, like another per-file index
14
+ entry: counted from the raw text, frontmatter and code included, without a parse. It reads that document's
15
+ bytes and nothing else.
16
+
17
+ Checks are plain functions of a database and a ref, like an inspection run over one file. Each asks for the
18
+ cheapest query that holds what it reads, so a check that needs only the frontmatter never pays for the full
19
+ parse, and every check reading the same query shares one computation of it. Their results are not cached; a
20
+ check runs again every time.
21
+
22
+ Nothing here records what a cached value read, so no dependency is tracked. Invalidation is written by hand instead,
23
+ the way the IDE drops per-file index entries on a file change event and resets structural caches on a project model
24
+ change. Reserved, not implemented: ``advance(snapshot) -> Database``, the next state. It would ``diff`` the two
25
+ snapshots and carry over each cached value the change set leaves valid: the frontmatter, the parse and the token
26
+ count of every document whose bytes did not change, and the model unless an entry under ``docs/`` was added or
27
+ deleted or a specification changed. That rule holds only while the frontmatter, the parse and the token count each
28
+ read their own document and the model reads no document, so keep them that way: data drawn from several documents
29
+ belongs in a new cache with its own rule.
30
+ """
31
+
32
+ from lorecraft.project.document import DocumentRef
33
+ from lorecraft.project.document import Repository as DocumentRepository
34
+ from lorecraft.project.syntax import FrontmatterNode, ParsedDocument, count_tokens, parse_document, parse_frontmatter
35
+ from lorecraft.project.workspace import WorkspaceModel, load_model
36
+ from lorecraft.vfs import Snapshot, VirtualFileSystem
37
+
38
+
39
+ class Database:
40
+ """The workspace model, the frontmatter, the parse trees and the token counts of one snapshot, each cached for
41
+ its lifetime."""
42
+
43
+ def __init__(self, snapshot: Snapshot) -> None:
44
+ """Index the snapshot for reading; performs no I/O and computes nothing yet."""
45
+ self._fs = VirtualFileSystem(snapshot)
46
+ self._documents = DocumentRepository(self._fs)
47
+ # `None` until the first `model()` call; a loaded model is never `None`, so the two cannot be confused.
48
+ self._model: WorkspaceModel | None = None
49
+ self._frontmatters: dict[DocumentRef, FrontmatterNode] = {}
50
+ self._parses: dict[DocumentRef, ParsedDocument] = {}
51
+ self._token_counts: dict[DocumentRef, int] = {}
52
+
53
+ def model(self) -> WorkspaceModel:
54
+ """The workspace model the snapshot declares, loaded on the first call.
55
+
56
+ A load that fails is not cached, so each call raises the same error again.
57
+
58
+ Raises:
59
+ ListSpecsError: If the specification directory cannot be listed.
60
+ ListCorpusDirectoriesError: If docs/ cannot be listed.
61
+ ListDocumentsError: If a corpus directory cannot be listed.
62
+ GetHeaderSchemaError: If any header schema cannot be read or decoded.
63
+ InvalidHeaderSchemaError: If any header schema is not a well-formed JSON Schema.
64
+ GetStructureSchemaError: If any structure specification cannot be read.
65
+ InvalidStructureSchemaError: If any structure specification is not JSON in the dialect, or states no
66
+ usable rules.
67
+ """
68
+ if self._model is None:
69
+ self._model = load_model(self._fs)
70
+ return self._model
71
+
72
+ def frontmatter(self, ref: DocumentRef) -> FrontmatterNode:
73
+ """The frontmatter of one document, parsed from the snapshot on the first call for its ref.
74
+
75
+ Cached apart from ``parse(ref)`` and never read from it, so the answer does not depend on which of the two
76
+ was asked first; ``parse_frontmatter`` guarantees the two agree.
77
+
78
+ A document that cannot be read is not cached, so each call raises the same error again.
79
+
80
+ Raises:
81
+ DocumentDecodeError: If the document's bytes are not UTF-8.
82
+ GetDocumentError: If the snapshot holds no regular file at the document's path.
83
+ """
84
+ decoded = self._frontmatters.get(ref)
85
+ if decoded is None:
86
+ text = self._documents.get_document(ref).text
87
+ decoded = parse_frontmatter(text)
88
+ self._frontmatters[ref] = decoded
89
+ return decoded
90
+
91
+ def parse(self, ref: DocumentRef) -> ParsedDocument:
92
+ """The parse tree of one document, parsed from the snapshot on the first call for its ref.
93
+
94
+ A document that cannot be read is not cached, so each call raises the same error again.
95
+
96
+ Raises:
97
+ DocumentDecodeError: If the document's bytes are not UTF-8.
98
+ GetDocumentError: If the snapshot holds no regular file at the document's path.
99
+ """
100
+ parsed = self._parses.get(ref)
101
+ if parsed is None:
102
+ text = self._documents.get_document(ref).text
103
+ parsed = parse_document(text)
104
+ self._parses[ref] = parsed
105
+ return parsed
106
+
107
+ def tokens(self, ref: DocumentRef) -> int:
108
+ """The tokens in one document's whole file, counted from the snapshot on the first call for its ref.
109
+
110
+ Cached apart from ``parse(ref)`` and never read from it: the count needs the raw text, not the tree, so
111
+ a check that needs only one of the two never pays for the other.
112
+
113
+ A document that cannot be read is not cached, so each call raises the same error again.
114
+
115
+ Raises:
116
+ DocumentDecodeError: If the document's bytes are not UTF-8.
117
+ GetDocumentError: If the snapshot holds no regular file at the document's path.
118
+ """
119
+ count = self._token_counts.get(ref)
120
+ if count is None:
121
+ text = self._documents.get_document(ref).text
122
+ count = count_tokens(text)
123
+ self._token_counts[ref] = count
124
+ return count
@@ -0,0 +1,122 @@
1
+ """Validate one document's frontmatter against the header schemas that govern it.
2
+
3
+ The check is pure: it takes the already decoded, already validated header schemas that govern a document, the
4
+ document's frontmatter node, and the filename and corpus the frontmatter is checked against, and returns violations.
5
+ It reads those and nothing else, so it is handed that node rather than the whole parse tree, and not the document's
6
+ path: the run that called it attaches that. Reading and parsing the document, choosing the schemas and deciding what
7
+ a decode failure means all happen above it, in ``checks.run`` and the database.
8
+ """
9
+
10
+ from dataclasses import dataclass
11
+ from typing import Final
12
+
13
+ from jsonschema import Draft202012Validator
14
+ from jsonschema.exceptions import ValidationError
15
+
16
+ from lorecraft.project.aspect import AspectFilename
17
+ from lorecraft.project.corpus import CorpusName
18
+ from lorecraft.project.schemas import HeaderAspect
19
+ from lorecraft.project.syntax import (
20
+ Frontmatter,
21
+ FrontmatterNode,
22
+ InvalidYamlFrontmatter,
23
+ LineNumber,
24
+ MissingFrontmatter,
25
+ NonMappingFrontmatter,
26
+ )
27
+
28
+ from .reporting import Violation
29
+
30
+ _FIRST_LINE: Final[LineNumber] = LineNumber(1)
31
+ """Where a violation with no more precise position is reported: a missing block, a missing key."""
32
+
33
+
34
+ @dataclass(frozen=True, slots=True)
35
+ class HeaderCheckResult:
36
+ """What the header check found in one document.
37
+
38
+ Attributes:
39
+ violations: In the order the check finds them; empty when the document conforms.
40
+ """
41
+
42
+ violations: tuple[Violation, ...]
43
+
44
+
45
+ def validate_header(
46
+ schemas: tuple[HeaderAspect, ...], *, frontmatter: FrontmatterNode, filename: AspectFilename, corpus: CorpusName
47
+ ) -> HeaderCheckResult:
48
+ """Check one document's frontmatter against the header schemas that govern it. Pure: raises nothing.
49
+
50
+ Args:
51
+ schemas: Applied in order; each violation names the schema's path. Empty means the document is
52
+ ungoverned, which yields no violations.
53
+ filename: The document's filename, which the frontmatter ``name`` must equal (rule
54
+ ``frontmatter.name-matches-filename``).
55
+ corpus: The document's corpus, which namespaces the rule of every schema violation.
56
+ """
57
+ if not schemas:
58
+ return HeaderCheckResult(violations=())
59
+
60
+ if isinstance(frontmatter, MissingFrontmatter):
61
+ return _one_violation('frontmatter.missing', 'no `---` delimited frontmatter block')
62
+ if isinstance(frontmatter, InvalidYamlFrontmatter):
63
+ return _one_violation('frontmatter.unparseable', f'frontmatter is not valid YAML: {frontmatter.detail}')
64
+ if isinstance(frontmatter, NonMappingFrontmatter):
65
+ return _one_violation('frontmatter.unparseable', 'frontmatter is not a YAML mapping')
66
+
67
+ violations: list[Violation] = []
68
+ expected_name = str(filename)
69
+ name = frontmatter.data.get('name')
70
+ if name != expected_name:
71
+ violations.append(
72
+ Violation(
73
+ line=_key_line(frontmatter, 'name'),
74
+ rule='frontmatter.name-matches-filename',
75
+ message=f'`name` is {name!r}; expected {expected_name!r}',
76
+ )
77
+ )
78
+
79
+ rule_namespace = str(corpus)
80
+ for aspect in schemas:
81
+ validator = Draft202012Validator(aspect.schema)
82
+ errors = sorted(
83
+ validator.iter_errors(frontmatter.data),
84
+ key=lambda error: (tuple(str(part) for part in error.path), error.message),
85
+ )
86
+ for error in errors:
87
+ for field in _violated_fields(error):
88
+ rule = f'{rule_namespace}.{field}' if field else f'{rule_namespace}.frontmatter'
89
+ violations.append(
90
+ Violation(
91
+ line=_key_line(frontmatter, field) if field else _FIRST_LINE,
92
+ rule=rule,
93
+ message=f'{error.message} (per {aspect.path})',
94
+ spec=aspect.path,
95
+ )
96
+ )
97
+
98
+ return HeaderCheckResult(violations=tuple(violations))
99
+
100
+
101
+ def _one_violation(rule: str, message: str) -> HeaderCheckResult:
102
+ """The result of a document whose frontmatter is unusable: one violation on its first line."""
103
+ return HeaderCheckResult(violations=(Violation(line=_FIRST_LINE, rule=rule, message=message),))
104
+
105
+
106
+ def _key_line(frontmatter: Frontmatter, key: str) -> LineNumber:
107
+ """The line a top-level key is written on, or line 1 when the frontmatter does not have it."""
108
+ line = frontmatter.key_line(key)
109
+ if line is None:
110
+ return _FIRST_LINE
111
+ return line
112
+
113
+
114
+ def _violated_fields(error: ValidationError) -> list[str]:
115
+ """Return the top-level fields implicated by one schema error."""
116
+ if error.path:
117
+ return [str(error.path[0])]
118
+
119
+ if error.validator == 'required' and isinstance(error.instance, dict):
120
+ return [str(field) for field in error.validator_value if field not in error.instance]
121
+
122
+ return ['']
@@ -0,0 +1,59 @@
1
+ """Shared findings and output formatting for document checks.
2
+
3
+ A check returns violations: where in the document a rule is broken, but not which document, since a check is a
4
+ pure function of what it reads and never needs the document's path. The run that checked a document knows which
5
+ it was, and ``Finding.at`` joins the two into a finding, the located form the output prints.
6
+ """
7
+
8
+ from dataclasses import dataclass
9
+ from typing import Self
10
+
11
+ from lorecraft.project.syntax import LineNumber
12
+ from lorecraft.vfs import RootRelativePath
13
+
14
+
15
+ @dataclass(frozen=True, slots=True)
16
+ class Violation:
17
+ """One broken rule in one document, without the document's path.
18
+
19
+ Attributes:
20
+ line: Where the violation is reported; line 1 when it concerns the whole document rather than one line.
21
+ rule: Stable identifier for the violated rule.
22
+ message: Human-readable explanation of the violation.
23
+ spec: The specification file that states the broken rule, or None, the default, for a rule the check
24
+ itself holds, such as a document that is not UTF-8.
25
+ """
26
+
27
+ line: LineNumber
28
+ rule: str
29
+ message: str
30
+ spec: RootRelativePath | None = None
31
+
32
+
33
+ @dataclass(frozen=True, slots=True)
34
+ class Finding:
35
+ """One validation finding, located in a repository document; frozen, so findings compare and hash by value.
36
+
37
+ Attributes:
38
+ path: The affected document; turned into text only where a finding is printed or serialised.
39
+ line: The line containing the finding.
40
+ rule: Stable identifier for the violated rule.
41
+ message: Human-readable explanation of the finding.
42
+ spec: The specification file that states the broken rule, or None for a rule the check itself holds.
43
+ """
44
+
45
+ path: RootRelativePath
46
+ line: LineNumber
47
+ rule: str
48
+ message: str
49
+ spec: RootRelativePath | None = None
50
+
51
+ @classmethod
52
+ def at(cls, path: RootRelativePath, violation: Violation) -> Self:
53
+ """The finding a violation is, in the document at ``path``."""
54
+ return cls(path=path, line=violation.line, rule=violation.rule, message=violation.message, spec=violation.spec)
55
+
56
+
57
+ def format_finding(finding: Finding) -> str:
58
+ """Format one finding for text output: ``<path>:<line>: [<rule>] <message>``."""
59
+ return f'{finding.path}:{finding.line}: [{finding.rule}] {finding.message}'
@@ -0,0 +1,230 @@
1
+ """Run a check over documents of one database.
2
+
3
+ A run asks the database for everything it reads: the model decides which aspects govern each document, and
4
+ the part of the document the check reads is what the pure check validates — the frontmatter for the header
5
+ check, the parse tree's headings for the structure check, the whole file's token count for the budget check.
6
+ Selecting which documents to check is the caller's business: a run checks the refs it is handed, in the order
7
+ given, and reads only the governed ones. Every check reports in the same shape, so the ``check`` commands print
8
+ every run the same way.
9
+ """
10
+
11
+ from dataclasses import dataclass
12
+
13
+ from lorecraft.project.document import DocumentDecodeError, DocumentRef
14
+ from lorecraft.project.schemas import StructureAspect
15
+ from lorecraft.project.syntax import FrontmatterNode, LineNumber, ParsedDocument
16
+
17
+ from .budget import validate_budget
18
+ from .database import Database
19
+ from .header import validate_header
20
+ from .reporting import Finding, Violation
21
+ from .structure import validate_structure
22
+
23
+
24
+ @dataclass(frozen=True, slots=True)
25
+ class DocumentReport:
26
+ """The outcome of checking one selected document.
27
+
28
+ Attributes:
29
+ ref: The document the report is about; its path is the report path.
30
+ governed: False when no specification governs the document for the check's aspect; the document was
31
+ then never parsed.
32
+ violations: What the check found, without the document's path; empty for an ungoverned document.
33
+ """
34
+
35
+ ref: DocumentRef
36
+ governed: bool
37
+ violations: tuple[Violation, ...]
38
+
39
+ def findings(self) -> tuple[Finding, ...]:
40
+ """Every violation, located in the report's document; empty exactly when the document is clean."""
41
+ return tuple(Finding.at(self.ref.path, violation) for violation in self.violations)
42
+
43
+
44
+ @dataclass(frozen=True, slots=True)
45
+ class CheckRun:
46
+ """One pass of one check over the selected documents: what the text and JSON printers consume.
47
+
48
+ Attributes:
49
+ reports: One per selected document, in the order the refs were given.
50
+ """
51
+
52
+ reports: tuple[DocumentReport, ...]
53
+
54
+ def findings(self) -> tuple[Finding, ...]:
55
+ """Every finding of every report, in report order; empty exactly when the run is clean."""
56
+ findings: list[Finding] = []
57
+ for report in self.reports:
58
+ findings.extend(report.findings())
59
+ return tuple(findings)
60
+
61
+
62
+ def run_header(database: Database, refs: tuple[DocumentRef, ...]) -> CheckRun:
63
+ """Check each ref against the header schemas that govern it, in the order given.
64
+
65
+ A governed document that is not UTF-8 carries the single violation ``frontmatter.undecodable`` at line 1.
66
+
67
+ Args:
68
+ database: The snapshot state the refs come from; its model decides which schemas govern each document.
69
+ refs: The documents to check; each must be one the database's model lists.
70
+
71
+ Raises:
72
+ GetDocumentError: If a governed document is missing from the snapshot; a decode failure is a finding.
73
+ ListSpecsError: If the model is not loaded yet and the specification directory cannot be listed.
74
+ ListCorpusDirectoriesError: If the model is not loaded yet and docs/ cannot be listed.
75
+ ListDocumentsError: If the model is not loaded yet and a corpus directory cannot be listed.
76
+ GetHeaderSchemaError: If the model is not loaded yet and a header schema cannot be read or decoded.
77
+ InvalidHeaderSchemaError: If the model is not loaded yet and a header schema is malformed.
78
+ GetStructureSchemaError: If the model is not loaded yet and a structure specification cannot be read.
79
+ InvalidStructureSchemaError: If the model is not loaded yet and a structure specification is malformed.
80
+ """
81
+ reports: list[DocumentReport] = []
82
+ for ref in refs:
83
+ aspects = database.model().governance(ref).header_schemas()
84
+ if not aspects:
85
+ reports.append(DocumentReport(ref, governed=False, violations=()))
86
+ continue
87
+ frontmatter = _frontmatter(database, ref)
88
+ if frontmatter is None:
89
+ reports.append(DocumentReport(ref, governed=True, violations=(_undecodable('frontmatter'),)))
90
+ continue
91
+ result = validate_header(aspects, frontmatter=frontmatter, filename=ref.filename, corpus=ref.corpus)
92
+ reports.append(DocumentReport(ref, governed=True, violations=result.violations))
93
+ return CheckRun(reports=tuple(reports))
94
+
95
+
96
+ def run_structure(database: Database, refs: tuple[DocumentRef, ...]) -> CheckRun:
97
+ """Check each ref against the structure specifications that govern it, in the order given.
98
+
99
+ A governed document that is not UTF-8 carries the single violation ``structure.undecodable`` at line 1.
100
+
101
+ Args:
102
+ database: The snapshot state the refs come from; its model decides which specifications govern each
103
+ document.
104
+ refs: The documents to check; each must be one the database's model lists.
105
+
106
+ Raises:
107
+ GetDocumentError: If a governed document is missing from the snapshot; a decode failure is a finding.
108
+ ListSpecsError: If the model is not loaded yet and the specification directory cannot be listed.
109
+ ListCorpusDirectoriesError: If the model is not loaded yet and docs/ cannot be listed.
110
+ ListDocumentsError: If the model is not loaded yet and a corpus directory cannot be listed.
111
+ GetHeaderSchemaError: If the model is not loaded yet and a header schema cannot be read or decoded.
112
+ InvalidHeaderSchemaError: If the model is not loaded yet and a header schema is malformed.
113
+ GetStructureSchemaError: If the model is not loaded yet and a structure specification cannot be read.
114
+ InvalidStructureSchemaError: If the model is not loaded yet and a structure specification is malformed.
115
+ """
116
+ reports: list[DocumentReport] = []
117
+ for ref in refs:
118
+ aspects = database.model().governance(ref).structure_specs()
119
+ if not aspects:
120
+ reports.append(DocumentReport(ref, governed=False, violations=()))
121
+ continue
122
+ document = _parse(database, ref)
123
+ if document is None:
124
+ reports.append(DocumentReport(ref, governed=True, violations=(_undecodable('structure'),)))
125
+ continue
126
+ result = validate_structure(aspects, headings=document.headings)
127
+ reports.append(DocumentReport(ref, governed=True, violations=result.violations))
128
+ return CheckRun(reports=tuple(reports))
129
+
130
+
131
+ def run_budget(database: Database, refs: tuple[DocumentRef, ...]) -> CheckRun:
132
+ """Check each ref's whole-file token count against the budgets that govern it, in the order given.
133
+
134
+ A document is governed when at least one of its structure specifications sets a ``tokens`` budget; one whose
135
+ specifications set none is ungoverned, and its text is never read. A governed document that is not UTF-8
136
+ carries the single violation ``budget.undecodable`` at line 1.
137
+
138
+ Args:
139
+ database: The snapshot state the refs come from; its model decides which specifications govern each
140
+ document.
141
+ refs: The documents to check; each must be one the database's model lists.
142
+
143
+ Raises:
144
+ GetDocumentError: If a governed document is missing from the snapshot; a decode failure is a finding.
145
+ ListSpecsError: If the model is not loaded yet and the specification directory cannot be listed.
146
+ ListCorpusDirectoriesError: If the model is not loaded yet and docs/ cannot be listed.
147
+ ListDocumentsError: If the model is not loaded yet and a corpus directory cannot be listed.
148
+ GetHeaderSchemaError: If the model is not loaded yet and a header schema cannot be read or decoded.
149
+ InvalidHeaderSchemaError: If the model is not loaded yet and a header schema is malformed.
150
+ GetStructureSchemaError: If the model is not loaded yet and a structure specification cannot be read.
151
+ InvalidStructureSchemaError: If the model is not loaded yet and a structure specification is malformed.
152
+ """
153
+ reports: list[DocumentReport] = []
154
+ for ref in refs:
155
+ aspects = _budgeted(database.model().governance(ref).structure_specs())
156
+ if not aspects:
157
+ reports.append(DocumentReport(ref, governed=False, violations=()))
158
+ continue
159
+ token_count = _tokens(database, ref)
160
+ if token_count is None:
161
+ reports.append(DocumentReport(ref, governed=True, violations=(_undecodable('budget'),)))
162
+ continue
163
+ result = validate_budget(aspects, token_count=token_count)
164
+ reports.append(DocumentReport(ref, governed=True, violations=result.violations))
165
+ return CheckRun(reports=tuple(reports))
166
+
167
+
168
+ def _budgeted(aspects: tuple[StructureAspect, ...]) -> tuple[StructureAspect, ...]:
169
+ """The structure aspects that set a ``tokens`` budget, in the order given."""
170
+ budgeted: list[StructureAspect] = []
171
+ for aspect in aspects:
172
+ if aspect.tokens is not None:
173
+ budgeted.append(aspect)
174
+ return tuple(budgeted)
175
+
176
+
177
+ def _frontmatter(database: Database, ref: DocumentRef) -> FrontmatterNode | None:
178
+ """The document's frontmatter node, or ``None`` when its bytes are not UTF-8.
179
+
180
+ Returns:
181
+ The frontmatter node. ``None`` is the degraded return ``_parse`` documents, for the same reason.
182
+
183
+ Raises:
184
+ GetDocumentError: If the document is missing from the snapshot; a decode failure is not raised.
185
+ """
186
+ try:
187
+ return database.frontmatter(ref)
188
+ except DocumentDecodeError:
189
+ return None
190
+
191
+
192
+ def _parse(database: Database, ref: DocumentRef) -> ParsedDocument | None:
193
+ """The document's parse tree, or ``None`` when its bytes are not UTF-8.
194
+
195
+ Returns:
196
+ The parse tree. ``None`` is the degraded return for bytes that are present but not UTF-8: those are on
197
+ the same side of the line as invalid YAML, since the document is wrong, so the caller reports a finding
198
+ rather than taking the exit-2 path an unreadable file takes.
199
+
200
+ Raises:
201
+ GetDocumentError: If the document is missing from the snapshot; a decode failure is not raised.
202
+ """
203
+ try:
204
+ return database.parse(ref)
205
+ except DocumentDecodeError:
206
+ return None
207
+
208
+
209
+ def _tokens(database: Database, ref: DocumentRef) -> int | None:
210
+ """The tokens in the document's whole file, or ``None`` when its bytes are not UTF-8.
211
+
212
+ Returns:
213
+ The token count. ``None`` is the degraded return ``_parse`` documents, for the same reason.
214
+
215
+ Raises:
216
+ GetDocumentError: If the document is missing from the snapshot; a decode failure is not raised.
217
+ """
218
+ try:
219
+ return database.tokens(ref)
220
+ except DocumentDecodeError:
221
+ return None
222
+
223
+
224
+ def _undecodable(rule_namespace: str) -> Violation:
225
+ """The violation a governed document that is not UTF-8 carries instead of the check's own."""
226
+ return Violation(
227
+ line=LineNumber(1),
228
+ rule=f'{rule_namespace}.undecodable',
229
+ message='document is not valid UTF-8',
230
+ )