constant-docs 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- constant_docs/__init__.py +18 -0
- constant_docs/__main__.py +15 -0
- constant_docs/api.py +968 -0
- constant_docs/auto.py +239 -0
- constant_docs/checks.py +494 -0
- constant_docs/cli.py +1085 -0
- constant_docs/config.py +1158 -0
- constant_docs/coverage.py +165 -0
- constant_docs/decisions.py +303 -0
- constant_docs/document.py +438 -0
- constant_docs/fingerprint.py +27 -0
- constant_docs/globs.py +154 -0
- constant_docs/guides/quickstart.md +124 -0
- constant_docs/guides/readme.md +198 -0
- constant_docs/house-style.md +70 -0
- constant_docs/index.py +210 -0
- constant_docs/kinds.py +304 -0
- constant_docs/paths.py +246 -0
- constant_docs/prompts/architecture.md +61 -0
- constant_docs/prompts/cli-reference.md +40 -0
- constant_docs/prompts/config-reference.md +38 -0
- constant_docs/prompts/errors.md +50 -0
- constant_docs/prompts/log.md +22 -0
- constant_docs/prompts/module.md +39 -0
- constant_docs/prompts/spec.md +68 -0
- constant_docs/state.py +273 -0
- constant_docs-0.4.0.dist-info/METADATA +380 -0
- constant_docs-0.4.0.dist-info/RECORD +31 -0
- constant_docs-0.4.0.dist-info/WHEEL +4 -0
- constant_docs-0.4.0.dist-info/entry_points.txt +3 -0
- constant_docs-0.4.0.dist-info/licenses/LICENSE +15 -0
constant_docs/checks.py
ADDED
|
@@ -0,0 +1,494 @@
|
|
|
1
|
+
"""Checks that read a document's body against the repository it describes.
|
|
2
|
+
|
|
3
|
+
Everything else in this tool compares a hash. That answers *"has the source
|
|
4
|
+
moved under this prose"* and nothing at all about whether the prose is right.
|
|
5
|
+
For most documents nothing better is available without parsing source, which is
|
|
6
|
+
a settled no.
|
|
7
|
+
|
|
8
|
+
For a few, something better is. This module holds those cases, and the rule for
|
|
9
|
+
admitting one is that the check must be **decidable from facts the tool already
|
|
10
|
+
has** — the configuration, the file list, a string scan — never from
|
|
11
|
+
understanding the prose.
|
|
12
|
+
|
|
13
|
+
## The architecture diagram
|
|
14
|
+
|
|
15
|
+
An architecture document carries a Mermaid diagram of how the parts fit
|
|
16
|
+
together. It is the model's reading of the code, not something derived from it:
|
|
17
|
+
this package parses no source, so the picture is prose in another form and
|
|
18
|
+
looks considerably more authoritative than it has earned.
|
|
19
|
+
|
|
20
|
+
One guard is available and costs nothing, and it catches the way these actually
|
|
21
|
+
rot — a module renamed or removed while the picture still shows the old name.
|
|
22
|
+
**Every module named in the diagram must exist in the configuration.**
|
|
23
|
+
|
|
24
|
+
A diagram also has to show things that are not modules: a harness, a
|
|
25
|
+
filesystem, a person. So the claim is marked rather than inferred — a **quoted**
|
|
26
|
+
node label names a module and is checked; an unquoted one is free structure.
|
|
27
|
+
That keeps an honest picture drawable and keeps the guard from becoming a
|
|
28
|
+
reason to draw a dishonest one.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import re
|
|
34
|
+
from collections.abc import Callable
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
from typing import Any
|
|
38
|
+
|
|
39
|
+
from constant_docs.document import DocumentCache, DocumentError
|
|
40
|
+
from constant_docs.kinds import ARCHITECTURE_CHECK, ERRORS_CHECK
|
|
41
|
+
from constant_docs.paths import doc_path_for
|
|
42
|
+
|
|
43
|
+
_FENCE_RE = re.compile(
|
|
44
|
+
r"^```([A-Za-z0-9_-]*)[^\S\n]*\n(.*?)^```", re.MULTILINE | re.DOTALL
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
# `cli["cli"]`, `cli("cli")`, `cli{"cli"}` — a quoted label in any node shape.
|
|
48
|
+
_QUOTED_LABEL_RE = re.compile(r"[\[\(\{]+\s*\"([^\"\n]*)\"\s*[\]\)\}]+")
|
|
49
|
+
|
|
50
|
+
_HEADER_RE = re.compile(r"^(graph|flowchart)\s+(TB|TD|BT|RL|LR)\s*$")
|
|
51
|
+
|
|
52
|
+
# Every edge form Mermaid's flowchart grammar spells with dashes or equals.
|
|
53
|
+
_EDGE_RE = re.compile(r"-\.->|-\.-|<-->|-{2,3}>|={2,3}>|-{2,3}|={2,3}")
|
|
54
|
+
_EDGE_LABEL_RE = re.compile(r"\|[^|\n]*\|")
|
|
55
|
+
_NODE_DECL_RE = re.compile(r"^\s*([A-Za-z_][\w-]*)\s*[\[\(\{]")
|
|
56
|
+
_BARE_ID_RE = re.compile(r"^([A-Za-z_][\w-]*)$")
|
|
57
|
+
|
|
58
|
+
# Mermaid's own vocabulary, which a bare-identifier match cannot tell from a
|
|
59
|
+
# node. `end` closes a `subgraph` and is a reserved word, so it can never be
|
|
60
|
+
# declared: a diagram using a subgraph was refused with advice — `end["..."]`
|
|
61
|
+
# — that is not writable in Mermaid at all.
|
|
62
|
+
_MERMAID_KEYWORDS = frozenset(
|
|
63
|
+
{"end", "subgraph", "graph", "flowchart", "direction", "click", "style"}
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
_BRACKETS = {"[": "]", "(": ")", "{": "}"}
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class DiagramError(DocumentError):
|
|
70
|
+
"""An architecture diagram names a module that is not configured.
|
|
71
|
+
|
|
72
|
+
A `DocumentError` because the consequence is the same as a missing required
|
|
73
|
+
heading: the body is refused and the document on disk is left as it was.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _mermaid_blocks(body: str) -> list[str]:
|
|
78
|
+
"""Return the contents of every fenced block tagged `mermaid`."""
|
|
79
|
+
return [
|
|
80
|
+
content
|
|
81
|
+
for language, content in _FENCE_RE.findall(body)
|
|
82
|
+
if language == "mermaid"
|
|
83
|
+
]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def diagram_labels(body: str) -> list[str]:
|
|
87
|
+
"""Return every quoted node label in *body*'s Mermaid diagrams, sorted.
|
|
88
|
+
|
|
89
|
+
Quoted because a diagram must be able to show a harness or a filesystem
|
|
90
|
+
without claiming it is a module. Marking the claim rather than inferring it
|
|
91
|
+
is what keeps the guard from making an honest picture undrawable.
|
|
92
|
+
"""
|
|
93
|
+
found: set[str] = set()
|
|
94
|
+
for block in _mermaid_blocks(body):
|
|
95
|
+
found.update(label.strip() for label in _QUOTED_LABEL_RE.findall(block))
|
|
96
|
+
return sorted(f for f in found if f)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def unknown_modules(labels: list[str], known: list[str]) -> list[str]:
|
|
100
|
+
"""Return the labels that name no configured module, sorted.
|
|
101
|
+
|
|
102
|
+
A label matches a module key exactly, or matches the last path segment of
|
|
103
|
+
exactly one key — full keys are long and an unreadable diagram is not a
|
|
104
|
+
diagram. A segment shared by two keys names neither, and reporting it is
|
|
105
|
+
the same reasoning that makes `probable_move` list every candidate rather
|
|
106
|
+
than pick one.
|
|
107
|
+
"""
|
|
108
|
+
keys = set(known)
|
|
109
|
+
by_segment: dict[str, int] = {}
|
|
110
|
+
for key in known:
|
|
111
|
+
by_segment[key.rstrip("/").split("/")[-1]] = (
|
|
112
|
+
by_segment.get(key.rstrip("/").split("/")[-1], 0) + 1
|
|
113
|
+
)
|
|
114
|
+
out = []
|
|
115
|
+
for raw in labels:
|
|
116
|
+
label = raw.strip().strip('"')
|
|
117
|
+
if label in keys:
|
|
118
|
+
continue
|
|
119
|
+
if by_segment.get(label) == 1:
|
|
120
|
+
continue
|
|
121
|
+
out.append(label)
|
|
122
|
+
return sorted(set(out))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def diagram_problems(body: str) -> list[str]:
|
|
126
|
+
"""Return the structural problems in *body*'s Mermaid diagrams.
|
|
127
|
+
|
|
128
|
+
Ours, not Mermaid's. A real parser is a JavaScript dependency and this
|
|
129
|
+
package takes one runtime dependency in total, so this checks the things
|
|
130
|
+
that actually stop a diagram rendering — a missing or unknown header, an
|
|
131
|
+
unclosed bracket or quote, an edge to a node that is never given a label —
|
|
132
|
+
and claims nothing beyond them.
|
|
133
|
+
"""
|
|
134
|
+
blocks = _mermaid_blocks(body)
|
|
135
|
+
if not blocks:
|
|
136
|
+
return [
|
|
137
|
+
(
|
|
138
|
+
"no Mermaid diagram: an architecture document carries one, in "
|
|
139
|
+
"a fenced block tagged `mermaid`"
|
|
140
|
+
)
|
|
141
|
+
]
|
|
142
|
+
|
|
143
|
+
problems: list[str] = []
|
|
144
|
+
for block in blocks:
|
|
145
|
+
lines = [line for line in block.splitlines() if line.strip()]
|
|
146
|
+
if not lines:
|
|
147
|
+
problems.append("empty Mermaid block")
|
|
148
|
+
continue
|
|
149
|
+
header = lines[0].strip()
|
|
150
|
+
if not _HEADER_RE.match(header):
|
|
151
|
+
problems.append(
|
|
152
|
+
f"diagram header {header!r} is not a `graph` or `flowchart` "
|
|
153
|
+
f"with a direction (TB, TD, BT, RL, LR)"
|
|
154
|
+
)
|
|
155
|
+
continue
|
|
156
|
+
|
|
157
|
+
declared: set[str] = set()
|
|
158
|
+
referenced: set[str] = set()
|
|
159
|
+
for line in lines[1:]:
|
|
160
|
+
problems.extend(_unbalanced(line))
|
|
161
|
+
stripped = _EDGE_LABEL_RE.sub(" ", line)
|
|
162
|
+
for fragment in _EDGE_RE.split(stripped):
|
|
163
|
+
fragment = fragment.strip()
|
|
164
|
+
if not fragment:
|
|
165
|
+
continue
|
|
166
|
+
declaration = _NODE_DECL_RE.match(fragment)
|
|
167
|
+
if declaration:
|
|
168
|
+
declared.add(declaration.group(1))
|
|
169
|
+
continue
|
|
170
|
+
bare = _BARE_ID_RE.match(fragment)
|
|
171
|
+
if bare and bare.group(1) not in _MERMAID_KEYWORDS:
|
|
172
|
+
referenced.add(bare.group(1))
|
|
173
|
+
|
|
174
|
+
for node in sorted(referenced - declared):
|
|
175
|
+
problems.append(
|
|
176
|
+
f"node {node!r} is joined by an edge but never given a label; "
|
|
177
|
+
f'declare it once as {node}["..."]'
|
|
178
|
+
)
|
|
179
|
+
return problems
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _unbalanced(line: str) -> list[str]:
|
|
183
|
+
"""Return a problem for *line* if its quotes or brackets do not close.
|
|
184
|
+
|
|
185
|
+
Counted rather than parsed, so a bracket inside a label reads as an
|
|
186
|
+
imbalance. That is a false positive on a label nobody should be writing,
|
|
187
|
+
and the alternative is a Mermaid grammar in this file.
|
|
188
|
+
"""
|
|
189
|
+
if line.count('"') % 2:
|
|
190
|
+
return [f"unbalanced quote: {line.strip()!r}"]
|
|
191
|
+
stack: list[str] = []
|
|
192
|
+
for char in line:
|
|
193
|
+
if char in _BRACKETS:
|
|
194
|
+
stack.append(_BRACKETS[char])
|
|
195
|
+
elif char in _BRACKETS.values() and (not stack or stack.pop() != char):
|
|
196
|
+
return [f"unbalanced bracket: {line.strip()!r}"]
|
|
197
|
+
if stack:
|
|
198
|
+
return [f"unbalanced bracket: {line.strip()!r}"]
|
|
199
|
+
return []
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def check_body(body: str, known: list[str]) -> list[str]:
|
|
203
|
+
"""Return every problem with an architecture body, structural then naming."""
|
|
204
|
+
problems = diagram_problems(body)
|
|
205
|
+
unknown = unknown_modules(diagram_labels(body), known)
|
|
206
|
+
if unknown:
|
|
207
|
+
problems.append(
|
|
208
|
+
"the diagram names "
|
|
209
|
+
+ ("a module" if len(unknown) == 1 else "modules")
|
|
210
|
+
+ " that the configuration does not have: "
|
|
211
|
+
+ ", ".join(unknown)
|
|
212
|
+
+ ". A picture still showing a renamed module is the way these "
|
|
213
|
+
"documents rot."
|
|
214
|
+
)
|
|
215
|
+
return problems
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def check_documents(
|
|
219
|
+
cfg: Any,
|
|
220
|
+
repo_root: Path,
|
|
221
|
+
check_name: str,
|
|
222
|
+
problems_for: Callable[[Any, str], list[str]],
|
|
223
|
+
cache: DocumentCache | None = None,
|
|
224
|
+
) -> list[str]:
|
|
225
|
+
"""Return the issues in every document whose kind declares *check_name*.
|
|
226
|
+
|
|
227
|
+
One sweep for both checks. The module loop, the missing-document skip, the
|
|
228
|
+
malformed-document skip and the message shape were written out twice and
|
|
229
|
+
had to agree by hand; a third checked kind would have written them a third
|
|
230
|
+
time, and a change to the shared policy had to be made in every copy or it
|
|
231
|
+
quietly applied to one kind only.
|
|
232
|
+
|
|
233
|
+
Selected by the kind's `check` field rather than by its name, so a project
|
|
234
|
+
that declares its own architecture kind keeps the guard instead of silently
|
|
235
|
+
losing it.
|
|
236
|
+
|
|
237
|
+
Reported rather than raised: `verify` names everything wrong in one run,
|
|
238
|
+
and a check that stopped at the first would need running as many times as
|
|
239
|
+
there are problems.
|
|
240
|
+
"""
|
|
241
|
+
read = (cache or DocumentCache()).load
|
|
242
|
+
issues: list[str] = []
|
|
243
|
+
for mod in cfg.modules:
|
|
244
|
+
kind = cfg.kinds.get(mod.kind)
|
|
245
|
+
if kind is None or kind.check != check_name:
|
|
246
|
+
continue
|
|
247
|
+
path = repo_root / doc_path_for(cfg, mod.key)
|
|
248
|
+
if not path.exists():
|
|
249
|
+
# A missing document is already reported as missing; saying so
|
|
250
|
+
# twice would make one problem look like two.
|
|
251
|
+
continue
|
|
252
|
+
try:
|
|
253
|
+
body = read(path).body
|
|
254
|
+
except (OSError, DocumentError, ValueError):
|
|
255
|
+
# Likewise: a malformed document is reported by the conformance
|
|
256
|
+
# check, which is where a reader will look for it.
|
|
257
|
+
continue
|
|
258
|
+
issues.extend(
|
|
259
|
+
f"{path.relative_to(repo_root).as_posix()} — {problem}"
|
|
260
|
+
for problem in problems_for(mod, body)
|
|
261
|
+
)
|
|
262
|
+
return issues
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def check_architecture(
|
|
266
|
+
cfg: Any, repo_root: Path, cache: DocumentCache | None = None
|
|
267
|
+
) -> list[str]:
|
|
268
|
+
"""Return the issues in every architecture document, for `verify`."""
|
|
269
|
+
known = [m.key for m in cfg.modules]
|
|
270
|
+
return check_documents(
|
|
271
|
+
cfg,
|
|
272
|
+
repo_root,
|
|
273
|
+
ARCHITECTURE_CHECK,
|
|
274
|
+
lambda _mod, body: check_body(body, known),
|
|
275
|
+
cache=cache,
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
# ---------------------------------------------------------------------------
|
|
280
|
+
# The error catalogue
|
|
281
|
+
# ---------------------------------------------------------------------------
|
|
282
|
+
#
|
|
283
|
+
# The one document this tool produces that can be verified in both directions:
|
|
284
|
+
# every error the source can raise appears in the catalogue, and every entry in
|
|
285
|
+
# the catalogue still exists in the source. Everything else it generates is
|
|
286
|
+
# unchecked prose that happens to be fresh.
|
|
287
|
+
#
|
|
288
|
+
# Neither direction needs parsing beyond a string scan — no AST, which stays a
|
|
289
|
+
# settled no — so it costs nothing and it holds. Built with the document rather
|
|
290
|
+
# than after it: a catalogue that is merely generated is prose, and a catalogue
|
|
291
|
+
# that is checked is a guarantee.
|
|
292
|
+
|
|
293
|
+
# `raise SomeError(` — the class name capitalised, which every exception in
|
|
294
|
+
# this codebase and the standard library is. A bare `raise` re-raises and emits
|
|
295
|
+
# no message of its own, so it is not a site.
|
|
296
|
+
_RAISE_RE = re.compile(r"\braise\s+([A-Z][A-Za-z0-9_]*)\s*\(")
|
|
297
|
+
|
|
298
|
+
# A quoted literal inside a message region. Prefixes cover f- and raw strings.
|
|
299
|
+
# A quoted literal inside a message region. Prefixes cover f- and raw strings.
|
|
300
|
+
# `\\.` before the delimiter alternative is what makes an escaped quote part of
|
|
301
|
+
# the string rather than its end: without it, `"expected \\"a\\" here"` was read
|
|
302
|
+
# as two literals and the catalogue was asked to match text no reader sees.
|
|
303
|
+
_LITERAL_RE = re.compile(
|
|
304
|
+
r"(?:[fFrRbB]{0,2})(\"\"\"|'''|\"|')((?:\\.|(?!\1)[\s\S])*?)\1"
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
# The escapes a message actually carries. Undone so the catalogue compares
|
|
308
|
+
# against what the reader sees, not the bytes the source spells it with.
|
|
309
|
+
_ESCAPE_RE = re.compile(r"\\([\\'\"])")
|
|
310
|
+
|
|
311
|
+
# The first cell of a table row, backticked. Marking the fragment by position
|
|
312
|
+
# rather than by syntax keeps every other backtick in the document — a flag, a
|
|
313
|
+
# filename, a class — from becoming a claim about the source.
|
|
314
|
+
_FRAGMENT_RE = re.compile(r"^\|\s*`([^`\n]+)`\s*\|", re.MULTILINE)
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
@dataclass(frozen=True)
|
|
318
|
+
class RaiseSite:
|
|
319
|
+
"""One `raise X(...)` with a literal message, and where it is."""
|
|
320
|
+
|
|
321
|
+
exception: str
|
|
322
|
+
message: str
|
|
323
|
+
line: int
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def raise_sites(text: str) -> list[RaiseSite]:
|
|
327
|
+
"""Return every raise site in *text* that carries a literal message.
|
|
328
|
+
|
|
329
|
+
A site whose message is built elsewhere — `raise SomeError(message)` — is
|
|
330
|
+
not returned: there is no string to catalogue, and demanding one would
|
|
331
|
+
make the check unsatisfiable rather than useful.
|
|
332
|
+
"""
|
|
333
|
+
out: list[RaiseSite] = []
|
|
334
|
+
for match in _RAISE_RE.finditer(text):
|
|
335
|
+
region = _balanced(text, match.end() - 1)
|
|
336
|
+
if region is None:
|
|
337
|
+
continue
|
|
338
|
+
literals = _message_literals(region)
|
|
339
|
+
if not literals:
|
|
340
|
+
continue
|
|
341
|
+
out.append(
|
|
342
|
+
RaiseSite(
|
|
343
|
+
exception=match.group(1),
|
|
344
|
+
# Joined, because a long message is written as adjacent string
|
|
345
|
+
# literals across several lines and a fragment may span the
|
|
346
|
+
# join. The space is what the reader sees at the seam.
|
|
347
|
+
message=" ".join(literals),
|
|
348
|
+
line=text.count("\n", 0, match.start()) + 1,
|
|
349
|
+
)
|
|
350
|
+
)
|
|
351
|
+
return out
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _message_literals(region: str) -> list[str]:
|
|
355
|
+
"""Return the literals in *region* that are part of the message.
|
|
356
|
+
|
|
357
|
+
A literal whose preceding non-space character is `=` is a keyword
|
|
358
|
+
argument — `key="modules"`, `from_=x` — and naming one in the catalogue
|
|
359
|
+
would let a row match on a word the user never sees. Excluding them also
|
|
360
|
+
means a raise whose message is built elsewhere and whose only literal is a
|
|
361
|
+
keyword argument is correctly not a site at all.
|
|
362
|
+
"""
|
|
363
|
+
out: list[str] = []
|
|
364
|
+
for match in _LITERAL_RE.finditer(region):
|
|
365
|
+
before = region[: match.start()].rstrip()
|
|
366
|
+
if before.endswith("=") and not before.endswith(("==", "!=", "<=", ">=")):
|
|
367
|
+
continue
|
|
368
|
+
out.append(_ESCAPE_RE.sub(r"\1", match.group(2)))
|
|
369
|
+
return out
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _balanced(text: str, open_paren: int) -> str | None:
|
|
373
|
+
"""Return the text between *open_paren* and its matching close, or None.
|
|
374
|
+
|
|
375
|
+
Counted rather than parsed, and quote-aware only enough to survive a
|
|
376
|
+
parenthesis inside a message.
|
|
377
|
+
"""
|
|
378
|
+
depth = 0
|
|
379
|
+
quote: str | None = None
|
|
380
|
+
i = open_paren
|
|
381
|
+
while i < len(text):
|
|
382
|
+
char = text[i]
|
|
383
|
+
if quote:
|
|
384
|
+
if char == "\\":
|
|
385
|
+
# Skip the backslash *and* what it escapes. Advancing by one
|
|
386
|
+
# left the escaped character to be examined on the next pass,
|
|
387
|
+
# so `\")` closed the string early and the argument region came
|
|
388
|
+
# back truncated — a message the source never raises.
|
|
389
|
+
i += 2
|
|
390
|
+
continue
|
|
391
|
+
if text.startswith(quote, i):
|
|
392
|
+
i += len(quote)
|
|
393
|
+
quote = None
|
|
394
|
+
continue
|
|
395
|
+
i += 1
|
|
396
|
+
continue
|
|
397
|
+
if char in "\"'":
|
|
398
|
+
quote = '"""' if text.startswith('"""', i) else char
|
|
399
|
+
quote = "'''" if text.startswith("'''", i) else quote
|
|
400
|
+
i += len(quote)
|
|
401
|
+
continue
|
|
402
|
+
if char == "(":
|
|
403
|
+
depth += 1
|
|
404
|
+
elif char == ")":
|
|
405
|
+
depth -= 1
|
|
406
|
+
if depth == 0:
|
|
407
|
+
return text[open_paren + 1 : i]
|
|
408
|
+
i += 1
|
|
409
|
+
return None
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def catalogue_fragments(body: str) -> list[str]:
|
|
413
|
+
"""Return the message fragments an errors document claims, sorted."""
|
|
414
|
+
return sorted({f.strip() for f in _FRAGMENT_RE.findall(body) if f.strip()})
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _normalise(text: str) -> str:
|
|
418
|
+
"""Collapse whitespace, so a message wrapped across lines still matches."""
|
|
419
|
+
return " ".join(text.split())
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def check_catalogue(body: str, sources: list[tuple[str, str]]) -> list[str]:
|
|
423
|
+
"""Return what the catalogue and the source disagree about, both ways.
|
|
424
|
+
|
|
425
|
+
*sources* is (repository-relative path, text) for every file the errors
|
|
426
|
+
document covers.
|
|
427
|
+
|
|
428
|
+
Reworded messages fail both directions at once and are reported twice,
|
|
429
|
+
deliberately: the entry that no longer matches anything and the site that
|
|
430
|
+
nothing covers are two separate things to fix, and collapsing them would
|
|
431
|
+
hide whichever the author did not think of.
|
|
432
|
+
"""
|
|
433
|
+
fragments = catalogue_fragments(body)
|
|
434
|
+
normalised = [(f, _normalise(f)) for f in fragments]
|
|
435
|
+
|
|
436
|
+
problems: list[str] = []
|
|
437
|
+
matched: set[str] = set()
|
|
438
|
+
|
|
439
|
+
for path, text in sources:
|
|
440
|
+
for site in raise_sites(text):
|
|
441
|
+
message = _normalise(site.message)
|
|
442
|
+
hits = [raw for raw, needle in normalised if needle in message]
|
|
443
|
+
if hits:
|
|
444
|
+
matched.update(hits)
|
|
445
|
+
continue
|
|
446
|
+
problems.append(
|
|
447
|
+
f"{path}:{site.line} raises {site.exception} with a message the "
|
|
448
|
+
f"catalogue does not carry: {_shorten(message)}"
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
for fragment in fragments:
|
|
452
|
+
if fragment not in matched:
|
|
453
|
+
problems.append(
|
|
454
|
+
f"the catalogue carries {fragment!r}, which no error in the "
|
|
455
|
+
f"source raises any more"
|
|
456
|
+
)
|
|
457
|
+
return problems
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def _shorten(text: str, limit: int = 80) -> str:
|
|
461
|
+
"""Return *text* short enough to read in a terminal, quoted."""
|
|
462
|
+
return repr(text if len(text) <= limit else text[: limit - 1] + "…")
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
class CatalogueError(DocumentError):
|
|
466
|
+
"""An error catalogue does not agree with the source, in either direction."""
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
def check_errors(
|
|
470
|
+
cfg: Any, repo_root: Path, cache: DocumentCache | None = None
|
|
471
|
+
) -> list[str]:
|
|
472
|
+
"""Return the completeness issues in every errors document, for `verify`."""
|
|
473
|
+
return check_documents(
|
|
474
|
+
cfg,
|
|
475
|
+
repo_root,
|
|
476
|
+
ERRORS_CHECK,
|
|
477
|
+
lambda mod, body: check_catalogue(body, sources_for(cfg, repo_root, mod.key)),
|
|
478
|
+
cache=cache,
|
|
479
|
+
)
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def sources_for(cfg: Any, repo_root: Path, module_key: str) -> list[tuple[str, str]]:
|
|
483
|
+
"""Return (relative path, text) for every file a module covers.
|
|
484
|
+
|
|
485
|
+
Only files that can be read as text, because a module glob may legitimately
|
|
486
|
+
take in an image or a lockfile and neither raises anything.
|
|
487
|
+
"""
|
|
488
|
+
out: list[tuple[str, str]] = []
|
|
489
|
+
for rel in cfg.module_files.get(module_key, []):
|
|
490
|
+
try:
|
|
491
|
+
out.append((rel.as_posix(), (repo_root / rel).read_text(encoding="utf-8")))
|
|
492
|
+
except (OSError, UnicodeDecodeError):
|
|
493
|
+
continue
|
|
494
|
+
return out
|