lorecraft 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lorecraft/__init__.py +5 -0
- lorecraft/__main__.py +6 -0
- lorecraft/checks/__init__.py +35 -0
- lorecraft/checks/budget.py +54 -0
- lorecraft/checks/database.py +124 -0
- lorecraft/checks/header.py +122 -0
- lorecraft/checks/reporting.py +59 -0
- lorecraft/checks/run.py +230 -0
- lorecraft/checks/structure.py +194 -0
- lorecraft/cli/__init__.py +9 -0
- lorecraft/cli/app.py +68 -0
- lorecraft/cli/check_run.py +194 -0
- lorecraft/cli/commands/__init__.py +8 -0
- lorecraft/cli/commands/check/__init__.py +86 -0
- lorecraft/cli/commands/check/budget.py +70 -0
- lorecraft/cli/commands/check/header.py +71 -0
- lorecraft/cli/commands/check/structure.py +70 -0
- lorecraft/cli/commands/inspect.py +52 -0
- lorecraft/cli/commands/version.py +22 -0
- lorecraft/cli/registry.py +111 -0
- lorecraft/cli/root.py +71 -0
- lorecraft/cli/select.py +102 -0
- lorecraft/cli/version.py +106 -0
- lorecraft/cli/workspace_tree.py +144 -0
- lorecraft/core/__init__.py +8 -0
- lorecraft/core/error.py +5 -0
- lorecraft/metadata.py +25 -0
- lorecraft/project/__init__.py +6 -0
- lorecraft/project/aspect/__init__.py +28 -0
- lorecraft/project/aspect/filename.py +53 -0
- lorecraft/project/aspect/name.py +96 -0
- lorecraft/project/aspect/namespace.py +90 -0
- lorecraft/project/corpus/__init__.py +15 -0
- lorecraft/project/corpus/name.py +96 -0
- lorecraft/project/document/__init__.py +23 -0
- lorecraft/project/document/ref.py +31 -0
- lorecraft/project/document/repo.py +147 -0
- lorecraft/project/layout.py +17 -0
- lorecraft/project/schemas/__init__.py +66 -0
- lorecraft/project/schemas/header.py +67 -0
- lorecraft/project/schemas/name.py +34 -0
- lorecraft/project/schemas/repo.py +146 -0
- lorecraft/project/schemas/spec_file.py +184 -0
- lorecraft/project/schemas/structure.py +280 -0
- lorecraft/project/schemas/structure_file.py +130 -0
- lorecraft/project/syntax/__init__.py +33 -0
- lorecraft/project/syntax/document.py +200 -0
- lorecraft/project/syntax/frontmatter.py +120 -0
- lorecraft/project/syntax/heading.py +34 -0
- lorecraft/project/syntax/o200k_base.tiktoken +199998 -0
- lorecraft/project/syntax/position.py +43 -0
- lorecraft/project/syntax/tokens.py +76 -0
- lorecraft/project/workspace/__init__.py +13 -0
- lorecraft/project/workspace/loader.py +192 -0
- lorecraft/project/workspace/model.py +214 -0
- lorecraft/vfs/__init__.py +47 -0
- lorecraft/vfs/changes.py +84 -0
- lorecraft/vfs/disk.py +323 -0
- lorecraft/vfs/path.py +107 -0
- lorecraft/vfs/snapshot.py +236 -0
- lorecraft/vfs/view.py +160 -0
- lorecraft-0.1.0.dist-info/METADATA +165 -0
- lorecraft-0.1.0.dist-info/RECORD +67 -0
- lorecraft-0.1.0.dist-info/WHEEL +4 -0
- lorecraft-0.1.0.dist-info/entry_points.txt +3 -0
- lorecraft-0.1.0.dist-info/licenses/LICENSE-APACHE +200 -0
- lorecraft-0.1.0.dist-info/licenses/LICENSE-MIT +21 -0
lorecraft/__init__.py
ADDED
lorecraft/__main__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Document checks: each validates one aspect of the documents a workspace model lists.
|
|
2
|
+
|
|
3
|
+
A check is pure over the one part of a document it reads: the header check over the frontmatter node, the
|
|
4
|
+
structure check over the headings, the budget check over the token count. The ``Database`` caches the model,
|
|
5
|
+
the frontmatter, the parse trees and the token counts of one snapshot, and every check reads through it, so each
|
|
6
|
+
part is computed once whichever checks read it; ``run`` asks the database for the part the check reads and hands
|
|
7
|
+
only that to the check. A check returns violations, which name no document; ``run`` files them under the
|
|
8
|
+
document's report, which locates them as findings. Nothing here prints: the ``check`` commands own the output
|
|
9
|
+
and the exit codes.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .budget import BudgetCheckResult, validate_budget
|
|
13
|
+
from .database import Database
|
|
14
|
+
from .header import HeaderCheckResult, validate_header
|
|
15
|
+
from .reporting import Finding, Violation, format_finding
|
|
16
|
+
from .run import CheckRun, DocumentReport, run_budget, run_header, run_structure
|
|
17
|
+
from .structure import StructureCheckResult, validate_structure
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
'Database',
|
|
21
|
+
'Finding',
|
|
22
|
+
'Violation',
|
|
23
|
+
'format_finding',
|
|
24
|
+
'HeaderCheckResult',
|
|
25
|
+
'validate_header',
|
|
26
|
+
'StructureCheckResult',
|
|
27
|
+
'validate_structure',
|
|
28
|
+
'BudgetCheckResult',
|
|
29
|
+
'validate_budget',
|
|
30
|
+
'DocumentReport',
|
|
31
|
+
'CheckRun',
|
|
32
|
+
'run_header',
|
|
33
|
+
'run_structure',
|
|
34
|
+
'run_budget',
|
|
35
|
+
]
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Check one document's whole-file token count against the token budgets that govern it.
|
|
2
|
+
|
|
3
|
+
The check is pure: it takes the already validated structure aspects that govern a document and the document's
|
|
4
|
+
token count, and returns violations. It reads the count and nothing else, so it is handed that number rather than
|
|
5
|
+
the text, and not the document's path: the run that called it attaches that. The budget is the global ``tokens``
|
|
6
|
+
key of a structure specification, but it is a check of its own: the structure check reads a document's parse tree,
|
|
7
|
+
and this one reads its raw text, frontmatter, code and tables included, since that is what loading it costs an
|
|
8
|
+
agent.
|
|
9
|
+
|
|
10
|
+
Each aspect is applied on its own. A namespace specification's budget does not replace the corpus one, so a
|
|
11
|
+
document governed by both must fit both, and a namespace can only tighten the corpus budget.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import Final
|
|
16
|
+
|
|
17
|
+
from lorecraft.project.schemas import StructureAspect
|
|
18
|
+
from lorecraft.project.syntax import LineNumber
|
|
19
|
+
|
|
20
|
+
from .reporting import Violation
|
|
21
|
+
|
|
22
|
+
_FIRST_LINE: Final[LineNumber] = LineNumber(1)
|
|
23
|
+
"""Where a budget violation is reported: it concerns the whole file, not one line of it."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class BudgetCheckResult:
|
|
28
|
+
"""What the budget check found in one document.
|
|
29
|
+
|
|
30
|
+
Attributes:
|
|
31
|
+
violations: One per aspect whose budget the document exceeds, in aspect order; empty when it fits all.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
violations: tuple[Violation, ...]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def validate_budget(aspects: tuple[StructureAspect, ...], *, token_count: int) -> BudgetCheckResult:
|
|
38
|
+
"""Check one document's token count against the budget of each structure aspect that sets one.
|
|
39
|
+
|
|
40
|
+
Every violation's message ends by naming the structure specification file that sets the budget, as the header
|
|
41
|
+
check names its schema: the number is in that file, not in the prose.
|
|
42
|
+
|
|
43
|
+
Args:
|
|
44
|
+
aspects: Applied each on its own; an aspect without a ``tokens`` budget is skipped. Empty means the
|
|
45
|
+
document is ungoverned, which yields no violations.
|
|
46
|
+
token_count: The ``o200k_base`` tokens in the document's whole file, as ``count_tokens`` counts them.
|
|
47
|
+
"""
|
|
48
|
+
violations: list[Violation] = []
|
|
49
|
+
for aspect in aspects:
|
|
50
|
+
if aspect.tokens is None or token_count <= aspect.tokens:
|
|
51
|
+
continue
|
|
52
|
+
message = f'{token_count} tokens; the budget is {aspect.tokens} (per {aspect.path.name})'
|
|
53
|
+
violations.append(Violation(line=_FIRST_LINE, rule='budget.tokens', message=message, spec=aspect.path))
|
|
54
|
+
return BudgetCheckResult(violations=tuple(violations))
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""One frozen state of a workspace: its snapshot, and the cached data every check reads through.
|
|
2
|
+
|
|
3
|
+
The layering follows the IntelliJ Platform's. The ``Snapshot`` plays the virtual file system: every byte a
|
|
4
|
+
check can see, and never changed once built. Above it sit two kinds of cached data, each computed on first
|
|
5
|
+
use and kept for as long as the database lives (pattern-memoization):
|
|
6
|
+
|
|
7
|
+
- ``model()``: the workspace model, like the IDE's project model. It reads the structure of the snapshot, the
|
|
8
|
+
specifications and the corpus directories, and no document's contents.
|
|
9
|
+
- ``frontmatter(ref)``: one document's frontmatter node, like a stub: the part of a file the IDE reads without
|
|
10
|
+
building its full syntax tree. It reads that document's bytes and nothing else.
|
|
11
|
+
- ``parse(ref)``: one document's parse tree, like a PSI file or a per-file index entry. It reads that
|
|
12
|
+
document's bytes and nothing else.
|
|
13
|
+
- ``tokens(ref)``: what one document's whole file costs an agent that loads it, like another per-file index
|
|
14
|
+
entry: counted from the raw text, frontmatter and code included, without a parse. It reads that document's
|
|
15
|
+
bytes and nothing else.
|
|
16
|
+
|
|
17
|
+
Checks are plain functions of a database and a ref, like an inspection run over one file. Each asks for the
|
|
18
|
+
cheapest query that holds what it reads, so a check that needs only the frontmatter never pays for the full
|
|
19
|
+
parse, and every check reading the same query shares one computation of it. Their results are not cached; a
|
|
20
|
+
check runs again every time.
|
|
21
|
+
|
|
22
|
+
Nothing here records what a cached value read, so no dependency is tracked. Invalidation is written by hand instead,
|
|
23
|
+
the way the IDE drops per-file index entries on a file change event and resets structural caches on a project model
|
|
24
|
+
change. Reserved, not implemented: ``advance(snapshot) -> Database``, the next state. It would ``diff`` the two
|
|
25
|
+
snapshots and carry over each cached value the change set leaves valid: the frontmatter, the parse and the token
|
|
26
|
+
count of every document whose bytes did not change, and the model unless an entry under ``docs/`` was added or
|
|
27
|
+
deleted or a specification changed. That rule holds only while the frontmatter, the parse and the token count each
|
|
28
|
+
read their own document and the model reads no document, so keep them that way: data drawn from several documents
|
|
29
|
+
belongs in a new cache with its own rule.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from lorecraft.project.document import DocumentRef
|
|
33
|
+
from lorecraft.project.document import Repository as DocumentRepository
|
|
34
|
+
from lorecraft.project.syntax import FrontmatterNode, ParsedDocument, count_tokens, parse_document, parse_frontmatter
|
|
35
|
+
from lorecraft.project.workspace import WorkspaceModel, load_model
|
|
36
|
+
from lorecraft.vfs import Snapshot, VirtualFileSystem
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Database:
|
|
40
|
+
"""The workspace model, the frontmatter, the parse trees and the token counts of one snapshot, each cached for
|
|
41
|
+
its lifetime."""
|
|
42
|
+
|
|
43
|
+
def __init__(self, snapshot: Snapshot) -> None:
|
|
44
|
+
"""Index the snapshot for reading; performs no I/O and computes nothing yet."""
|
|
45
|
+
self._fs = VirtualFileSystem(snapshot)
|
|
46
|
+
self._documents = DocumentRepository(self._fs)
|
|
47
|
+
# `None` until the first `model()` call; a loaded model is never `None`, so the two cannot be confused.
|
|
48
|
+
self._model: WorkspaceModel | None = None
|
|
49
|
+
self._frontmatters: dict[DocumentRef, FrontmatterNode] = {}
|
|
50
|
+
self._parses: dict[DocumentRef, ParsedDocument] = {}
|
|
51
|
+
self._token_counts: dict[DocumentRef, int] = {}
|
|
52
|
+
|
|
53
|
+
def model(self) -> WorkspaceModel:
|
|
54
|
+
"""The workspace model the snapshot declares, loaded on the first call.
|
|
55
|
+
|
|
56
|
+
A load that fails is not cached, so each call raises the same error again.
|
|
57
|
+
|
|
58
|
+
Raises:
|
|
59
|
+
ListSpecsError: If the specification directory cannot be listed.
|
|
60
|
+
ListCorpusDirectoriesError: If docs/ cannot be listed.
|
|
61
|
+
ListDocumentsError: If a corpus directory cannot be listed.
|
|
62
|
+
GetHeaderSchemaError: If any header schema cannot be read or decoded.
|
|
63
|
+
InvalidHeaderSchemaError: If any header schema is not a well-formed JSON Schema.
|
|
64
|
+
GetStructureSchemaError: If any structure specification cannot be read.
|
|
65
|
+
InvalidStructureSchemaError: If any structure specification is not JSON in the dialect, or states no
|
|
66
|
+
usable rules.
|
|
67
|
+
"""
|
|
68
|
+
if self._model is None:
|
|
69
|
+
self._model = load_model(self._fs)
|
|
70
|
+
return self._model
|
|
71
|
+
|
|
72
|
+
def frontmatter(self, ref: DocumentRef) -> FrontmatterNode:
|
|
73
|
+
"""The frontmatter of one document, parsed from the snapshot on the first call for its ref.
|
|
74
|
+
|
|
75
|
+
Cached apart from ``parse(ref)`` and never read from it, so the answer does not depend on which of the two
|
|
76
|
+
was asked first; ``parse_frontmatter`` guarantees the two agree.
|
|
77
|
+
|
|
78
|
+
A document that cannot be read is not cached, so each call raises the same error again.
|
|
79
|
+
|
|
80
|
+
Raises:
|
|
81
|
+
DocumentDecodeError: If the document's bytes are not UTF-8.
|
|
82
|
+
GetDocumentError: If the snapshot holds no regular file at the document's path.
|
|
83
|
+
"""
|
|
84
|
+
decoded = self._frontmatters.get(ref)
|
|
85
|
+
if decoded is None:
|
|
86
|
+
text = self._documents.get_document(ref).text
|
|
87
|
+
decoded = parse_frontmatter(text)
|
|
88
|
+
self._frontmatters[ref] = decoded
|
|
89
|
+
return decoded
|
|
90
|
+
|
|
91
|
+
def parse(self, ref: DocumentRef) -> ParsedDocument:
|
|
92
|
+
"""The parse tree of one document, parsed from the snapshot on the first call for its ref.
|
|
93
|
+
|
|
94
|
+
A document that cannot be read is not cached, so each call raises the same error again.
|
|
95
|
+
|
|
96
|
+
Raises:
|
|
97
|
+
DocumentDecodeError: If the document's bytes are not UTF-8.
|
|
98
|
+
GetDocumentError: If the snapshot holds no regular file at the document's path.
|
|
99
|
+
"""
|
|
100
|
+
parsed = self._parses.get(ref)
|
|
101
|
+
if parsed is None:
|
|
102
|
+
text = self._documents.get_document(ref).text
|
|
103
|
+
parsed = parse_document(text)
|
|
104
|
+
self._parses[ref] = parsed
|
|
105
|
+
return parsed
|
|
106
|
+
|
|
107
|
+
def tokens(self, ref: DocumentRef) -> int:
|
|
108
|
+
"""The tokens in one document's whole file, counted from the snapshot on the first call for its ref.
|
|
109
|
+
|
|
110
|
+
Cached apart from ``parse(ref)`` and never read from it: the count needs the raw text, not the tree, so
|
|
111
|
+
a check that needs only one of the two never pays for the other.
|
|
112
|
+
|
|
113
|
+
A document that cannot be read is not cached, so each call raises the same error again.
|
|
114
|
+
|
|
115
|
+
Raises:
|
|
116
|
+
DocumentDecodeError: If the document's bytes are not UTF-8.
|
|
117
|
+
GetDocumentError: If the snapshot holds no regular file at the document's path.
|
|
118
|
+
"""
|
|
119
|
+
count = self._token_counts.get(ref)
|
|
120
|
+
if count is None:
|
|
121
|
+
text = self._documents.get_document(ref).text
|
|
122
|
+
count = count_tokens(text)
|
|
123
|
+
self._token_counts[ref] = count
|
|
124
|
+
return count
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Validate one document's frontmatter against the header schemas that govern it.
|
|
2
|
+
|
|
3
|
+
The check is pure: it takes the already decoded, already validated header schemas that govern a document, the
|
|
4
|
+
document's frontmatter node, and the filename and corpus the frontmatter is checked against, and returns violations.
|
|
5
|
+
It reads those and nothing else, so it is handed that node rather than the whole parse tree, and not the document's
|
|
6
|
+
path: the run that called it attaches that. Reading and parsing the document, choosing the schemas and deciding what
|
|
7
|
+
a decode failure means all happen above it, in ``checks.run`` and the database.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Final
|
|
12
|
+
|
|
13
|
+
from jsonschema import Draft202012Validator
|
|
14
|
+
from jsonschema.exceptions import ValidationError
|
|
15
|
+
|
|
16
|
+
from lorecraft.project.aspect import AspectFilename
|
|
17
|
+
from lorecraft.project.corpus import CorpusName
|
|
18
|
+
from lorecraft.project.schemas import HeaderAspect
|
|
19
|
+
from lorecraft.project.syntax import (
|
|
20
|
+
Frontmatter,
|
|
21
|
+
FrontmatterNode,
|
|
22
|
+
InvalidYamlFrontmatter,
|
|
23
|
+
LineNumber,
|
|
24
|
+
MissingFrontmatter,
|
|
25
|
+
NonMappingFrontmatter,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
from .reporting import Violation
|
|
29
|
+
|
|
30
|
+
_FIRST_LINE: Final[LineNumber] = LineNumber(1)
|
|
31
|
+
"""Where a violation with no more precise position is reported: a missing block, a missing key."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(frozen=True, slots=True)
|
|
35
|
+
class HeaderCheckResult:
|
|
36
|
+
"""What the header check found in one document.
|
|
37
|
+
|
|
38
|
+
Attributes:
|
|
39
|
+
violations: In the order the check finds them; empty when the document conforms.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
violations: tuple[Violation, ...]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def validate_header(
|
|
46
|
+
schemas: tuple[HeaderAspect, ...], *, frontmatter: FrontmatterNode, filename: AspectFilename, corpus: CorpusName
|
|
47
|
+
) -> HeaderCheckResult:
|
|
48
|
+
"""Check one document's frontmatter against the header schemas that govern it. Pure: raises nothing.
|
|
49
|
+
|
|
50
|
+
Args:
|
|
51
|
+
schemas: Applied in order; each violation names the schema's path. Empty means the document is
|
|
52
|
+
ungoverned, which yields no violations.
|
|
53
|
+
filename: The document's filename, which the frontmatter ``name`` must equal (rule
|
|
54
|
+
``frontmatter.name-matches-filename``).
|
|
55
|
+
corpus: The document's corpus, which namespaces the rule of every schema violation.
|
|
56
|
+
"""
|
|
57
|
+
if not schemas:
|
|
58
|
+
return HeaderCheckResult(violations=())
|
|
59
|
+
|
|
60
|
+
if isinstance(frontmatter, MissingFrontmatter):
|
|
61
|
+
return _one_violation('frontmatter.missing', 'no `---` delimited frontmatter block')
|
|
62
|
+
if isinstance(frontmatter, InvalidYamlFrontmatter):
|
|
63
|
+
return _one_violation('frontmatter.unparseable', f'frontmatter is not valid YAML: {frontmatter.detail}')
|
|
64
|
+
if isinstance(frontmatter, NonMappingFrontmatter):
|
|
65
|
+
return _one_violation('frontmatter.unparseable', 'frontmatter is not a YAML mapping')
|
|
66
|
+
|
|
67
|
+
violations: list[Violation] = []
|
|
68
|
+
expected_name = str(filename)
|
|
69
|
+
name = frontmatter.data.get('name')
|
|
70
|
+
if name != expected_name:
|
|
71
|
+
violations.append(
|
|
72
|
+
Violation(
|
|
73
|
+
line=_key_line(frontmatter, 'name'),
|
|
74
|
+
rule='frontmatter.name-matches-filename',
|
|
75
|
+
message=f'`name` is {name!r}; expected {expected_name!r}',
|
|
76
|
+
)
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
rule_namespace = str(corpus)
|
|
80
|
+
for aspect in schemas:
|
|
81
|
+
validator = Draft202012Validator(aspect.schema)
|
|
82
|
+
errors = sorted(
|
|
83
|
+
validator.iter_errors(frontmatter.data),
|
|
84
|
+
key=lambda error: (tuple(str(part) for part in error.path), error.message),
|
|
85
|
+
)
|
|
86
|
+
for error in errors:
|
|
87
|
+
for field in _violated_fields(error):
|
|
88
|
+
rule = f'{rule_namespace}.{field}' if field else f'{rule_namespace}.frontmatter'
|
|
89
|
+
violations.append(
|
|
90
|
+
Violation(
|
|
91
|
+
line=_key_line(frontmatter, field) if field else _FIRST_LINE,
|
|
92
|
+
rule=rule,
|
|
93
|
+
message=f'{error.message} (per {aspect.path})',
|
|
94
|
+
spec=aspect.path,
|
|
95
|
+
)
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
return HeaderCheckResult(violations=tuple(violations))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _one_violation(rule: str, message: str) -> HeaderCheckResult:
|
|
102
|
+
"""The result of a document whose frontmatter is unusable: one violation on its first line."""
|
|
103
|
+
return HeaderCheckResult(violations=(Violation(line=_FIRST_LINE, rule=rule, message=message),))
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _key_line(frontmatter: Frontmatter, key: str) -> LineNumber:
|
|
107
|
+
"""The line a top-level key is written on, or line 1 when the frontmatter does not have it."""
|
|
108
|
+
line = frontmatter.key_line(key)
|
|
109
|
+
if line is None:
|
|
110
|
+
return _FIRST_LINE
|
|
111
|
+
return line
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _violated_fields(error: ValidationError) -> list[str]:
|
|
115
|
+
"""Return the top-level fields implicated by one schema error."""
|
|
116
|
+
if error.path:
|
|
117
|
+
return [str(error.path[0])]
|
|
118
|
+
|
|
119
|
+
if error.validator == 'required' and isinstance(error.instance, dict):
|
|
120
|
+
return [str(field) for field in error.validator_value if field not in error.instance]
|
|
121
|
+
|
|
122
|
+
return ['']
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Shared findings and output formatting for document checks.
|
|
2
|
+
|
|
3
|
+
A check returns violations: where in the document a rule is broken, but not which document, since a check is a
|
|
4
|
+
pure function of what it reads and never needs the document's path. The run that checked a document knows which
|
|
5
|
+
it was, and ``Finding.at`` joins the two into a finding, the located form the output prints.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Self
|
|
10
|
+
|
|
11
|
+
from lorecraft.project.syntax import LineNumber
|
|
12
|
+
from lorecraft.vfs import RootRelativePath
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True, slots=True)
|
|
16
|
+
class Violation:
|
|
17
|
+
"""One broken rule in one document, without the document's path.
|
|
18
|
+
|
|
19
|
+
Attributes:
|
|
20
|
+
line: Where the violation is reported; line 1 when it concerns the whole document rather than one line.
|
|
21
|
+
rule: Stable identifier for the violated rule.
|
|
22
|
+
message: Human-readable explanation of the violation.
|
|
23
|
+
spec: The specification file that states the broken rule, or None, the default, for a rule the check
|
|
24
|
+
itself holds, such as a document that is not UTF-8.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
line: LineNumber
|
|
28
|
+
rule: str
|
|
29
|
+
message: str
|
|
30
|
+
spec: RootRelativePath | None = None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True, slots=True)
|
|
34
|
+
class Finding:
|
|
35
|
+
"""One validation finding, located in a repository document; frozen, so findings compare and hash by value.
|
|
36
|
+
|
|
37
|
+
Attributes:
|
|
38
|
+
path: The affected document; turned into text only where a finding is printed or serialised.
|
|
39
|
+
line: The line containing the finding.
|
|
40
|
+
rule: Stable identifier for the violated rule.
|
|
41
|
+
message: Human-readable explanation of the finding.
|
|
42
|
+
spec: The specification file that states the broken rule, or None for a rule the check itself holds.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
path: RootRelativePath
|
|
46
|
+
line: LineNumber
|
|
47
|
+
rule: str
|
|
48
|
+
message: str
|
|
49
|
+
spec: RootRelativePath | None = None
|
|
50
|
+
|
|
51
|
+
@classmethod
|
|
52
|
+
def at(cls, path: RootRelativePath, violation: Violation) -> Self:
|
|
53
|
+
"""The finding a violation is, in the document at ``path``."""
|
|
54
|
+
return cls(path=path, line=violation.line, rule=violation.rule, message=violation.message, spec=violation.spec)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def format_finding(finding: Finding) -> str:
|
|
58
|
+
"""Format one finding for text output: ``<path>:<line>: [<rule>] <message>``."""
|
|
59
|
+
return f'{finding.path}:{finding.line}: [{finding.rule}] {finding.message}'
|
lorecraft/checks/run.py
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""Run a check over documents of one database.
|
|
2
|
+
|
|
3
|
+
A run asks the database for everything it reads: the model decides which aspects govern each document, and
|
|
4
|
+
the part of the document the check reads is what the pure check validates — the frontmatter for the header
|
|
5
|
+
check, the parse tree's headings for the structure check, the whole file's token count for the budget check.
|
|
6
|
+
Selecting which documents to check is the caller's business: a run checks the refs it is handed, in the order
|
|
7
|
+
given, and reads only the governed ones. Every check reports in the same shape, so the ``check`` commands print
|
|
8
|
+
every run the same way.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
from lorecraft.project.document import DocumentDecodeError, DocumentRef
|
|
14
|
+
from lorecraft.project.schemas import StructureAspect
|
|
15
|
+
from lorecraft.project.syntax import FrontmatterNode, LineNumber, ParsedDocument
|
|
16
|
+
|
|
17
|
+
from .budget import validate_budget
|
|
18
|
+
from .database import Database
|
|
19
|
+
from .header import validate_header
|
|
20
|
+
from .reporting import Finding, Violation
|
|
21
|
+
from .structure import validate_structure
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True, slots=True)
|
|
25
|
+
class DocumentReport:
|
|
26
|
+
"""The outcome of checking one selected document.
|
|
27
|
+
|
|
28
|
+
Attributes:
|
|
29
|
+
ref: The document the report is about; its path is the report path.
|
|
30
|
+
governed: False when no specification governs the document for the check's aspect; the document was
|
|
31
|
+
then never parsed.
|
|
32
|
+
violations: What the check found, without the document's path; empty for an ungoverned document.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
ref: DocumentRef
|
|
36
|
+
governed: bool
|
|
37
|
+
violations: tuple[Violation, ...]
|
|
38
|
+
|
|
39
|
+
def findings(self) -> tuple[Finding, ...]:
|
|
40
|
+
"""Every violation, located in the report's document; empty exactly when the document is clean."""
|
|
41
|
+
return tuple(Finding.at(self.ref.path, violation) for violation in self.violations)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True, slots=True)
|
|
45
|
+
class CheckRun:
|
|
46
|
+
"""One pass of one check over the selected documents: what the text and JSON printers consume.
|
|
47
|
+
|
|
48
|
+
Attributes:
|
|
49
|
+
reports: One per selected document, in the order the refs were given.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
reports: tuple[DocumentReport, ...]
|
|
53
|
+
|
|
54
|
+
def findings(self) -> tuple[Finding, ...]:
|
|
55
|
+
"""Every finding of every report, in report order; empty exactly when the run is clean."""
|
|
56
|
+
findings: list[Finding] = []
|
|
57
|
+
for report in self.reports:
|
|
58
|
+
findings.extend(report.findings())
|
|
59
|
+
return tuple(findings)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def run_header(database: Database, refs: tuple[DocumentRef, ...]) -> CheckRun:
|
|
63
|
+
"""Check each ref against the header schemas that govern it, in the order given.
|
|
64
|
+
|
|
65
|
+
A governed document that is not UTF-8 carries the single violation ``frontmatter.undecodable`` at line 1.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
database: The snapshot state the refs come from; its model decides which schemas govern each document.
|
|
69
|
+
refs: The documents to check; each must be one the database's model lists.
|
|
70
|
+
|
|
71
|
+
Raises:
|
|
72
|
+
GetDocumentError: If a governed document is missing from the snapshot; a decode failure is a finding.
|
|
73
|
+
ListSpecsError: If the model is not loaded yet and the specification directory cannot be listed.
|
|
74
|
+
ListCorpusDirectoriesError: If the model is not loaded yet and docs/ cannot be listed.
|
|
75
|
+
ListDocumentsError: If the model is not loaded yet and a corpus directory cannot be listed.
|
|
76
|
+
GetHeaderSchemaError: If the model is not loaded yet and a header schema cannot be read or decoded.
|
|
77
|
+
InvalidHeaderSchemaError: If the model is not loaded yet and a header schema is malformed.
|
|
78
|
+
GetStructureSchemaError: If the model is not loaded yet and a structure specification cannot be read.
|
|
79
|
+
InvalidStructureSchemaError: If the model is not loaded yet and a structure specification is malformed.
|
|
80
|
+
"""
|
|
81
|
+
reports: list[DocumentReport] = []
|
|
82
|
+
for ref in refs:
|
|
83
|
+
aspects = database.model().governance(ref).header_schemas()
|
|
84
|
+
if not aspects:
|
|
85
|
+
reports.append(DocumentReport(ref, governed=False, violations=()))
|
|
86
|
+
continue
|
|
87
|
+
frontmatter = _frontmatter(database, ref)
|
|
88
|
+
if frontmatter is None:
|
|
89
|
+
reports.append(DocumentReport(ref, governed=True, violations=(_undecodable('frontmatter'),)))
|
|
90
|
+
continue
|
|
91
|
+
result = validate_header(aspects, frontmatter=frontmatter, filename=ref.filename, corpus=ref.corpus)
|
|
92
|
+
reports.append(DocumentReport(ref, governed=True, violations=result.violations))
|
|
93
|
+
return CheckRun(reports=tuple(reports))
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def run_structure(database: Database, refs: tuple[DocumentRef, ...]) -> CheckRun:
|
|
97
|
+
"""Check each ref against the structure specifications that govern it, in the order given.
|
|
98
|
+
|
|
99
|
+
A governed document that is not UTF-8 carries the single violation ``structure.undecodable`` at line 1.
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
database: The snapshot state the refs come from; its model decides which specifications govern each
|
|
103
|
+
document.
|
|
104
|
+
refs: The documents to check; each must be one the database's model lists.
|
|
105
|
+
|
|
106
|
+
Raises:
|
|
107
|
+
GetDocumentError: If a governed document is missing from the snapshot; a decode failure is a finding.
|
|
108
|
+
ListSpecsError: If the model is not loaded yet and the specification directory cannot be listed.
|
|
109
|
+
ListCorpusDirectoriesError: If the model is not loaded yet and docs/ cannot be listed.
|
|
110
|
+
ListDocumentsError: If the model is not loaded yet and a corpus directory cannot be listed.
|
|
111
|
+
GetHeaderSchemaError: If the model is not loaded yet and a header schema cannot be read or decoded.
|
|
112
|
+
InvalidHeaderSchemaError: If the model is not loaded yet and a header schema is malformed.
|
|
113
|
+
GetStructureSchemaError: If the model is not loaded yet and a structure specification cannot be read.
|
|
114
|
+
InvalidStructureSchemaError: If the model is not loaded yet and a structure specification is malformed.
|
|
115
|
+
"""
|
|
116
|
+
reports: list[DocumentReport] = []
|
|
117
|
+
for ref in refs:
|
|
118
|
+
aspects = database.model().governance(ref).structure_specs()
|
|
119
|
+
if not aspects:
|
|
120
|
+
reports.append(DocumentReport(ref, governed=False, violations=()))
|
|
121
|
+
continue
|
|
122
|
+
document = _parse(database, ref)
|
|
123
|
+
if document is None:
|
|
124
|
+
reports.append(DocumentReport(ref, governed=True, violations=(_undecodable('structure'),)))
|
|
125
|
+
continue
|
|
126
|
+
result = validate_structure(aspects, headings=document.headings)
|
|
127
|
+
reports.append(DocumentReport(ref, governed=True, violations=result.violations))
|
|
128
|
+
return CheckRun(reports=tuple(reports))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def run_budget(database: Database, refs: tuple[DocumentRef, ...]) -> CheckRun:
|
|
132
|
+
"""Check each ref's whole-file token count against the budgets that govern it, in the order given.
|
|
133
|
+
|
|
134
|
+
A document is governed when at least one of its structure specifications sets a ``tokens`` budget; one whose
|
|
135
|
+
specifications set none is ungoverned, and its text is never read. A governed document that is not UTF-8
|
|
136
|
+
carries the single violation ``budget.undecodable`` at line 1.
|
|
137
|
+
|
|
138
|
+
Args:
|
|
139
|
+
database: The snapshot state the refs come from; its model decides which specifications govern each
|
|
140
|
+
document.
|
|
141
|
+
refs: The documents to check; each must be one the database's model lists.
|
|
142
|
+
|
|
143
|
+
Raises:
|
|
144
|
+
GetDocumentError: If a governed document is missing from the snapshot; a decode failure is a finding.
|
|
145
|
+
ListSpecsError: If the model is not loaded yet and the specification directory cannot be listed.
|
|
146
|
+
ListCorpusDirectoriesError: If the model is not loaded yet and docs/ cannot be listed.
|
|
147
|
+
ListDocumentsError: If the model is not loaded yet and a corpus directory cannot be listed.
|
|
148
|
+
GetHeaderSchemaError: If the model is not loaded yet and a header schema cannot be read or decoded.
|
|
149
|
+
InvalidHeaderSchemaError: If the model is not loaded yet and a header schema is malformed.
|
|
150
|
+
GetStructureSchemaError: If the model is not loaded yet and a structure specification cannot be read.
|
|
151
|
+
InvalidStructureSchemaError: If the model is not loaded yet and a structure specification is malformed.
|
|
152
|
+
"""
|
|
153
|
+
reports: list[DocumentReport] = []
|
|
154
|
+
for ref in refs:
|
|
155
|
+
aspects = _budgeted(database.model().governance(ref).structure_specs())
|
|
156
|
+
if not aspects:
|
|
157
|
+
reports.append(DocumentReport(ref, governed=False, violations=()))
|
|
158
|
+
continue
|
|
159
|
+
token_count = _tokens(database, ref)
|
|
160
|
+
if token_count is None:
|
|
161
|
+
reports.append(DocumentReport(ref, governed=True, violations=(_undecodable('budget'),)))
|
|
162
|
+
continue
|
|
163
|
+
result = validate_budget(aspects, token_count=token_count)
|
|
164
|
+
reports.append(DocumentReport(ref, governed=True, violations=result.violations))
|
|
165
|
+
return CheckRun(reports=tuple(reports))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _budgeted(aspects: tuple[StructureAspect, ...]) -> tuple[StructureAspect, ...]:
|
|
169
|
+
"""The structure aspects that set a ``tokens`` budget, in the order given."""
|
|
170
|
+
budgeted: list[StructureAspect] = []
|
|
171
|
+
for aspect in aspects:
|
|
172
|
+
if aspect.tokens is not None:
|
|
173
|
+
budgeted.append(aspect)
|
|
174
|
+
return tuple(budgeted)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _frontmatter(database: Database, ref: DocumentRef) -> FrontmatterNode | None:
|
|
178
|
+
"""The document's frontmatter node, or ``None`` when its bytes are not UTF-8.
|
|
179
|
+
|
|
180
|
+
Returns:
|
|
181
|
+
The frontmatter node. ``None`` is the degraded return ``_parse`` documents, for the same reason.
|
|
182
|
+
|
|
183
|
+
Raises:
|
|
184
|
+
GetDocumentError: If the document is missing from the snapshot; a decode failure is not raised.
|
|
185
|
+
"""
|
|
186
|
+
try:
|
|
187
|
+
return database.frontmatter(ref)
|
|
188
|
+
except DocumentDecodeError:
|
|
189
|
+
return None
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _parse(database: Database, ref: DocumentRef) -> ParsedDocument | None:
|
|
193
|
+
"""The document's parse tree, or ``None`` when its bytes are not UTF-8.
|
|
194
|
+
|
|
195
|
+
Returns:
|
|
196
|
+
The parse tree. ``None`` is the degraded return for bytes that are present but not UTF-8: those are on
|
|
197
|
+
the same side of the line as invalid YAML, since the document is wrong, so the caller reports a finding
|
|
198
|
+
rather than taking the exit-2 path an unreadable file takes.
|
|
199
|
+
|
|
200
|
+
Raises:
|
|
201
|
+
GetDocumentError: If the document is missing from the snapshot; a decode failure is not raised.
|
|
202
|
+
"""
|
|
203
|
+
try:
|
|
204
|
+
return database.parse(ref)
|
|
205
|
+
except DocumentDecodeError:
|
|
206
|
+
return None
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _tokens(database: Database, ref: DocumentRef) -> int | None:
|
|
210
|
+
"""The tokens in the document's whole file, or ``None`` when its bytes are not UTF-8.
|
|
211
|
+
|
|
212
|
+
Returns:
|
|
213
|
+
The token count. ``None`` is the degraded return ``_parse`` documents, for the same reason.
|
|
214
|
+
|
|
215
|
+
Raises:
|
|
216
|
+
GetDocumentError: If the document is missing from the snapshot; a decode failure is not raised.
|
|
217
|
+
"""
|
|
218
|
+
try:
|
|
219
|
+
return database.tokens(ref)
|
|
220
|
+
except DocumentDecodeError:
|
|
221
|
+
return None
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _undecodable(rule_namespace: str) -> Violation:
|
|
225
|
+
"""The violation a governed document that is not UTF-8 carries instead of the check's own."""
|
|
226
|
+
return Violation(
|
|
227
|
+
line=LineNumber(1),
|
|
228
|
+
rule=f'{rule_namespace}.undecodable',
|
|
229
|
+
message='document is not valid UTF-8',
|
|
230
|
+
)
|