constant-docs 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,494 @@
1
+ """Checks that read a document's body against the repository it describes.
2
+
3
+ Everything else in this tool compares a hash. That answers *"has the source
4
+ moved under this prose"* and nothing at all about whether the prose is right.
5
+ For most documents nothing better is available without parsing source, which is
6
+ a settled no.
7
+
8
+ For a few, something better is. This module holds those cases, and the rule for
9
+ admitting one is that the check must be **decidable from facts the tool already
10
+ has** — the configuration, the file list, a string scan — never from
11
+ understanding the prose.
12
+
13
+ ## The architecture diagram
14
+
15
+ An architecture document carries a Mermaid diagram of how the parts fit
16
+ together. It is the model's reading of the code, not something derived from it:
17
+ this package parses no source, so the picture is prose in another form and
18
+ looks considerably more authoritative than it has earned.
19
+
20
+ One guard is available and costs nothing, and it catches the way these actually
21
+ rot — a module renamed or removed while the picture still shows the old name.
22
+ **Every module named in the diagram must exist in the configuration.**
23
+
24
+ A diagram also has to show things that are not modules: a harness, a
25
+ filesystem, a person. So the claim is marked rather than inferred — a **quoted**
26
+ node label names a module and is checked; an unquoted one is free structure.
27
+ That keeps an honest picture drawable and keeps the guard from becoming a
28
+ reason to draw a dishonest one.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import re
34
+ from collections.abc import Callable
35
+ from dataclasses import dataclass
36
+ from pathlib import Path
37
+ from typing import Any
38
+
39
+ from constant_docs.document import DocumentCache, DocumentError
40
+ from constant_docs.kinds import ARCHITECTURE_CHECK, ERRORS_CHECK
41
+ from constant_docs.paths import doc_path_for
42
+
43
+ _FENCE_RE = re.compile(
44
+ r"^```([A-Za-z0-9_-]*)[^\S\n]*\n(.*?)^```", re.MULTILINE | re.DOTALL
45
+ )
46
+
47
+ # `cli["cli"]`, `cli("cli")`, `cli{"cli"}` — a quoted label in any node shape.
48
+ _QUOTED_LABEL_RE = re.compile(r"[\[\(\{]+\s*\"([^\"\n]*)\"\s*[\]\)\}]+")
49
+
50
+ _HEADER_RE = re.compile(r"^(graph|flowchart)\s+(TB|TD|BT|RL|LR)\s*$")
51
+
52
+ # Every edge form Mermaid's flowchart grammar spells with dashes or equals.
53
+ _EDGE_RE = re.compile(r"-\.->|-\.-|<-->|-{2,3}>|={2,3}>|-{2,3}|={2,3}")
54
+ _EDGE_LABEL_RE = re.compile(r"\|[^|\n]*\|")
55
+ _NODE_DECL_RE = re.compile(r"^\s*([A-Za-z_][\w-]*)\s*[\[\(\{]")
56
+ _BARE_ID_RE = re.compile(r"^([A-Za-z_][\w-]*)$")
57
+
58
+ # Mermaid's own vocabulary, which a bare-identifier match cannot tell from a
59
+ # node. `end` closes a `subgraph` and is a reserved word, so it can never be
60
+ # declared: a diagram using a subgraph was refused with advice — `end["..."]`
61
+ # — that is not writable in Mermaid at all.
62
+ _MERMAID_KEYWORDS = frozenset(
63
+ {"end", "subgraph", "graph", "flowchart", "direction", "click", "style"}
64
+ )
65
+
66
+ _BRACKETS = {"[": "]", "(": ")", "{": "}"}
67
+
68
+
69
+ class DiagramError(DocumentError):
70
+ """An architecture diagram names a module that is not configured.
71
+
72
+ A `DocumentError` because the consequence is the same as a missing required
73
+ heading: the body is refused and the document on disk is left as it was.
74
+ """
75
+
76
+
77
+ def _mermaid_blocks(body: str) -> list[str]:
78
+ """Return the contents of every fenced block tagged `mermaid`."""
79
+ return [
80
+ content
81
+ for language, content in _FENCE_RE.findall(body)
82
+ if language == "mermaid"
83
+ ]
84
+
85
+
86
+ def diagram_labels(body: str) -> list[str]:
87
+ """Return every quoted node label in *body*'s Mermaid diagrams, sorted.
88
+
89
+ Quoted because a diagram must be able to show a harness or a filesystem
90
+ without claiming it is a module. Marking the claim rather than inferring it
91
+ is what keeps the guard from making an honest picture undrawable.
92
+ """
93
+ found: set[str] = set()
94
+ for block in _mermaid_blocks(body):
95
+ found.update(label.strip() for label in _QUOTED_LABEL_RE.findall(block))
96
+ return sorted(f for f in found if f)
97
+
98
+
99
+ def unknown_modules(labels: list[str], known: list[str]) -> list[str]:
100
+ """Return the labels that name no configured module, sorted.
101
+
102
+ A label matches a module key exactly, or matches the last path segment of
103
+ exactly one key — full keys are long and an unreadable diagram is not a
104
+ diagram. A segment shared by two keys names neither, and reporting it is
105
+ the same reasoning that makes `probable_move` list every candidate rather
106
+ than pick one.
107
+ """
108
+ keys = set(known)
109
+ by_segment: dict[str, int] = {}
110
+ for key in known:
111
+ by_segment[key.rstrip("/").split("/")[-1]] = (
112
+ by_segment.get(key.rstrip("/").split("/")[-1], 0) + 1
113
+ )
114
+ out = []
115
+ for raw in labels:
116
+ label = raw.strip().strip('"')
117
+ if label in keys:
118
+ continue
119
+ if by_segment.get(label) == 1:
120
+ continue
121
+ out.append(label)
122
+ return sorted(set(out))
123
+
124
+
125
+ def diagram_problems(body: str) -> list[str]:
126
+ """Return the structural problems in *body*'s Mermaid diagrams.
127
+
128
+ Ours, not Mermaid's. A real parser is a JavaScript dependency and this
129
+ package takes one runtime dependency in total, so this checks the things
130
+ that actually stop a diagram rendering — a missing or unknown header, an
131
+ unclosed bracket or quote, an edge to a node that is never given a label —
132
+ and claims nothing beyond them.
133
+ """
134
+ blocks = _mermaid_blocks(body)
135
+ if not blocks:
136
+ return [
137
+ (
138
+ "no Mermaid diagram: an architecture document carries one, in "
139
+ "a fenced block tagged `mermaid`"
140
+ )
141
+ ]
142
+
143
+ problems: list[str] = []
144
+ for block in blocks:
145
+ lines = [line for line in block.splitlines() if line.strip()]
146
+ if not lines:
147
+ problems.append("empty Mermaid block")
148
+ continue
149
+ header = lines[0].strip()
150
+ if not _HEADER_RE.match(header):
151
+ problems.append(
152
+ f"diagram header {header!r} is not a `graph` or `flowchart` "
153
+ f"with a direction (TB, TD, BT, RL, LR)"
154
+ )
155
+ continue
156
+
157
+ declared: set[str] = set()
158
+ referenced: set[str] = set()
159
+ for line in lines[1:]:
160
+ problems.extend(_unbalanced(line))
161
+ stripped = _EDGE_LABEL_RE.sub(" ", line)
162
+ for fragment in _EDGE_RE.split(stripped):
163
+ fragment = fragment.strip()
164
+ if not fragment:
165
+ continue
166
+ declaration = _NODE_DECL_RE.match(fragment)
167
+ if declaration:
168
+ declared.add(declaration.group(1))
169
+ continue
170
+ bare = _BARE_ID_RE.match(fragment)
171
+ if bare and bare.group(1) not in _MERMAID_KEYWORDS:
172
+ referenced.add(bare.group(1))
173
+
174
+ for node in sorted(referenced - declared):
175
+ problems.append(
176
+ f"node {node!r} is joined by an edge but never given a label; "
177
+ f'declare it once as {node}["..."]'
178
+ )
179
+ return problems
180
+
181
+
182
+ def _unbalanced(line: str) -> list[str]:
183
+ """Return a problem for *line* if its quotes or brackets do not close.
184
+
185
+ Counted rather than parsed, so a bracket inside a label reads as an
186
+ imbalance. That is a false positive on a label nobody should be writing,
187
+ and the alternative is a Mermaid grammar in this file.
188
+ """
189
+ if line.count('"') % 2:
190
+ return [f"unbalanced quote: {line.strip()!r}"]
191
+ stack: list[str] = []
192
+ for char in line:
193
+ if char in _BRACKETS:
194
+ stack.append(_BRACKETS[char])
195
+ elif char in _BRACKETS.values() and (not stack or stack.pop() != char):
196
+ return [f"unbalanced bracket: {line.strip()!r}"]
197
+ if stack:
198
+ return [f"unbalanced bracket: {line.strip()!r}"]
199
+ return []
200
+
201
+
202
+ def check_body(body: str, known: list[str]) -> list[str]:
203
+ """Return every problem with an architecture body, structural then naming."""
204
+ problems = diagram_problems(body)
205
+ unknown = unknown_modules(diagram_labels(body), known)
206
+ if unknown:
207
+ problems.append(
208
+ "the diagram names "
209
+ + ("a module" if len(unknown) == 1 else "modules")
210
+ + " that the configuration does not have: "
211
+ + ", ".join(unknown)
212
+ + ". A picture still showing a renamed module is the way these "
213
+ "documents rot."
214
+ )
215
+ return problems
216
+
217
+
218
+ def check_documents(
219
+ cfg: Any,
220
+ repo_root: Path,
221
+ check_name: str,
222
+ problems_for: Callable[[Any, str], list[str]],
223
+ cache: DocumentCache | None = None,
224
+ ) -> list[str]:
225
+ """Return the issues in every document whose kind declares *check_name*.
226
+
227
+ One sweep for both checks. The module loop, the missing-document skip, the
228
+ malformed-document skip and the message shape were written out twice and
229
+ had to agree by hand; a third checked kind would have written them a third
230
+ time, and a change to the shared policy had to be made in every copy or it
231
+ quietly applied to one kind only.
232
+
233
+ Selected by the kind's `check` field rather than by its name, so a project
234
+ that declares its own architecture kind keeps the guard instead of silently
235
+ losing it.
236
+
237
+ Reported rather than raised: `verify` names everything wrong in one run,
238
+ and a check that stopped at the first would need running as many times as
239
+ there are problems.
240
+ """
241
+ read = (cache or DocumentCache()).load
242
+ issues: list[str] = []
243
+ for mod in cfg.modules:
244
+ kind = cfg.kinds.get(mod.kind)
245
+ if kind is None or kind.check != check_name:
246
+ continue
247
+ path = repo_root / doc_path_for(cfg, mod.key)
248
+ if not path.exists():
249
+ # A missing document is already reported as missing; saying so
250
+ # twice would make one problem look like two.
251
+ continue
252
+ try:
253
+ body = read(path).body
254
+ except (OSError, DocumentError, ValueError):
255
+ # Likewise: a malformed document is reported by the conformance
256
+ # check, which is where a reader will look for it.
257
+ continue
258
+ issues.extend(
259
+ f"{path.relative_to(repo_root).as_posix()} — {problem}"
260
+ for problem in problems_for(mod, body)
261
+ )
262
+ return issues
263
+
264
+
265
+ def check_architecture(
266
+ cfg: Any, repo_root: Path, cache: DocumentCache | None = None
267
+ ) -> list[str]:
268
+ """Return the issues in every architecture document, for `verify`."""
269
+ known = [m.key for m in cfg.modules]
270
+ return check_documents(
271
+ cfg,
272
+ repo_root,
273
+ ARCHITECTURE_CHECK,
274
+ lambda _mod, body: check_body(body, known),
275
+ cache=cache,
276
+ )
277
+
278
+
279
+ # ---------------------------------------------------------------------------
280
+ # The error catalogue
281
+ # ---------------------------------------------------------------------------
282
+ #
283
+ # The one document this tool produces that can be verified in both directions:
284
+ # every error the source can raise appears in the catalogue, and every entry in
285
+ # the catalogue still exists in the source. Everything else it generates is
286
+ # unchecked prose that happens to be fresh.
287
+ #
288
+ # Neither direction needs parsing beyond a string scan — no AST, which stays a
289
+ # settled no — so it costs nothing and it holds. Built with the document rather
290
+ # than after it: a catalogue that is merely generated is prose, and a catalogue
291
+ # that is checked is a guarantee.
292
+
293
+ # `raise SomeError(` — the class name capitalised, which every exception in
294
+ # this codebase and the standard library is. A bare `raise` re-raises and emits
295
+ # no message of its own, so it is not a site.
296
+ _RAISE_RE = re.compile(r"\braise\s+([A-Z][A-Za-z0-9_]*)\s*\(")
297
+
298
+ # A quoted literal inside a message region. Prefixes cover f- and raw strings.
299
+ # A quoted literal inside a message region. Prefixes cover f- and raw strings.
300
+ # `\\.` before the delimiter alternative is what makes an escaped quote part of
301
+ # the string rather than its end: without it, `"expected \\"a\\" here"` was read
302
+ # as two literals and the catalogue was asked to match text no reader sees.
303
+ _LITERAL_RE = re.compile(
304
+ r"(?:[fFrRbB]{0,2})(\"\"\"|'''|\"|')((?:\\.|(?!\1)[\s\S])*?)\1"
305
+ )
306
+
307
+ # The escapes a message actually carries. Undone so the catalogue compares
308
+ # against what the reader sees, not the bytes the source spells it with.
309
+ _ESCAPE_RE = re.compile(r"\\([\\'\"])")
310
+
311
+ # The first cell of a table row, backticked. Marking the fragment by position
312
+ # rather than by syntax keeps every other backtick in the document — a flag, a
313
+ # filename, a class — from becoming a claim about the source.
314
+ _FRAGMENT_RE = re.compile(r"^\|\s*`([^`\n]+)`\s*\|", re.MULTILINE)
315
+
316
+
317
+ @dataclass(frozen=True)
318
+ class RaiseSite:
319
+ """One `raise X(...)` with a literal message, and where it is."""
320
+
321
+ exception: str
322
+ message: str
323
+ line: int
324
+
325
+
326
+ def raise_sites(text: str) -> list[RaiseSite]:
327
+ """Return every raise site in *text* that carries a literal message.
328
+
329
+ A site whose message is built elsewhere — `raise SomeError(message)` — is
330
+ not returned: there is no string to catalogue, and demanding one would
331
+ make the check unsatisfiable rather than useful.
332
+ """
333
+ out: list[RaiseSite] = []
334
+ for match in _RAISE_RE.finditer(text):
335
+ region = _balanced(text, match.end() - 1)
336
+ if region is None:
337
+ continue
338
+ literals = _message_literals(region)
339
+ if not literals:
340
+ continue
341
+ out.append(
342
+ RaiseSite(
343
+ exception=match.group(1),
344
+ # Joined, because a long message is written as adjacent string
345
+ # literals across several lines and a fragment may span the
346
+ # join. The space is what the reader sees at the seam.
347
+ message=" ".join(literals),
348
+ line=text.count("\n", 0, match.start()) + 1,
349
+ )
350
+ )
351
+ return out
352
+
353
+
354
+ def _message_literals(region: str) -> list[str]:
355
+ """Return the literals in *region* that are part of the message.
356
+
357
+ A literal whose preceding non-space character is `=` is a keyword
358
+ argument — `key="modules"`, `from_=x` — and naming one in the catalogue
359
+ would let a row match on a word the user never sees. Excluding them also
360
+ means a raise whose message is built elsewhere and whose only literal is a
361
+ keyword argument is correctly not a site at all.
362
+ """
363
+ out: list[str] = []
364
+ for match in _LITERAL_RE.finditer(region):
365
+ before = region[: match.start()].rstrip()
366
+ if before.endswith("=") and not before.endswith(("==", "!=", "<=", ">=")):
367
+ continue
368
+ out.append(_ESCAPE_RE.sub(r"\1", match.group(2)))
369
+ return out
370
+
371
+
372
+ def _balanced(text: str, open_paren: int) -> str | None:
373
+ """Return the text between *open_paren* and its matching close, or None.
374
+
375
+ Counted rather than parsed, and quote-aware only enough to survive a
376
+ parenthesis inside a message.
377
+ """
378
+ depth = 0
379
+ quote: str | None = None
380
+ i = open_paren
381
+ while i < len(text):
382
+ char = text[i]
383
+ if quote:
384
+ if char == "\\":
385
+ # Skip the backslash *and* what it escapes. Advancing by one
386
+ # left the escaped character to be examined on the next pass,
387
+ # so `\")` closed the string early and the argument region came
388
+ # back truncated — a message the source never raises.
389
+ i += 2
390
+ continue
391
+ if text.startswith(quote, i):
392
+ i += len(quote)
393
+ quote = None
394
+ continue
395
+ i += 1
396
+ continue
397
+ if char in "\"'":
398
+ quote = '"""' if text.startswith('"""', i) else char
399
+ quote = "'''" if text.startswith("'''", i) else quote
400
+ i += len(quote)
401
+ continue
402
+ if char == "(":
403
+ depth += 1
404
+ elif char == ")":
405
+ depth -= 1
406
+ if depth == 0:
407
+ return text[open_paren + 1 : i]
408
+ i += 1
409
+ return None
410
+
411
+
412
+ def catalogue_fragments(body: str) -> list[str]:
413
+ """Return the message fragments an errors document claims, sorted."""
414
+ return sorted({f.strip() for f in _FRAGMENT_RE.findall(body) if f.strip()})
415
+
416
+
417
+ def _normalise(text: str) -> str:
418
+ """Collapse whitespace, so a message wrapped across lines still matches."""
419
+ return " ".join(text.split())
420
+
421
+
422
+ def check_catalogue(body: str, sources: list[tuple[str, str]]) -> list[str]:
423
+ """Return what the catalogue and the source disagree about, both ways.
424
+
425
+ *sources* is (repository-relative path, text) for every file the errors
426
+ document covers.
427
+
428
+ Reworded messages fail both directions at once and are reported twice,
429
+ deliberately: the entry that no longer matches anything and the site that
430
+ nothing covers are two separate things to fix, and collapsing them would
431
+ hide whichever the author did not think of.
432
+ """
433
+ fragments = catalogue_fragments(body)
434
+ normalised = [(f, _normalise(f)) for f in fragments]
435
+
436
+ problems: list[str] = []
437
+ matched: set[str] = set()
438
+
439
+ for path, text in sources:
440
+ for site in raise_sites(text):
441
+ message = _normalise(site.message)
442
+ hits = [raw for raw, needle in normalised if needle in message]
443
+ if hits:
444
+ matched.update(hits)
445
+ continue
446
+ problems.append(
447
+ f"{path}:{site.line} raises {site.exception} with a message the "
448
+ f"catalogue does not carry: {_shorten(message)}"
449
+ )
450
+
451
+ for fragment in fragments:
452
+ if fragment not in matched:
453
+ problems.append(
454
+ f"the catalogue carries {fragment!r}, which no error in the "
455
+ f"source raises any more"
456
+ )
457
+ return problems
458
+
459
+
460
+ def _shorten(text: str, limit: int = 80) -> str:
461
+ """Return *text* short enough to read in a terminal, quoted."""
462
+ return repr(text if len(text) <= limit else text[: limit - 1] + "…")
463
+
464
+
465
+ class CatalogueError(DocumentError):
466
+ """An error catalogue does not agree with the source, in either direction."""
467
+
468
+
469
+ def check_errors(
470
+ cfg: Any, repo_root: Path, cache: DocumentCache | None = None
471
+ ) -> list[str]:
472
+ """Return the completeness issues in every errors document, for `verify`."""
473
+ return check_documents(
474
+ cfg,
475
+ repo_root,
476
+ ERRORS_CHECK,
477
+ lambda mod, body: check_catalogue(body, sources_for(cfg, repo_root, mod.key)),
478
+ cache=cache,
479
+ )
480
+
481
+
482
+ def sources_for(cfg: Any, repo_root: Path, module_key: str) -> list[tuple[str, str]]:
483
+ """Return (relative path, text) for every file a module covers.
484
+
485
+ Only files that can be read as text, because a module glob may legitimately
486
+ take in an image or a lockfile and neither raises anything.
487
+ """
488
+ out: list[tuple[str, str]] = []
489
+ for rel in cfg.module_files.get(module_key, []):
490
+ try:
491
+ out.append((rel.as_posix(), (repo_root / rel).read_text(encoding="utf-8")))
492
+ except (OSError, UnicodeDecodeError):
493
+ continue
494
+ return out